{"1024-x-1024/50-steps/bedrock/amazon.nova-canvas-v1:0":{"mode":"image_generation","base_model":"nova-canvas","max_input_tokens":2600,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2600}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1024-x-1024/50-steps/stability.stable-diffusion-xl-v1":{"mode":"image_generation","base_model":"stable-diffusion-xl-v1","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1024-x-1024/dall-e-2":{"mode":"image_generation","base_model":"dall-e-2","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1024-x-1024/max-steps/stability.stable-diffusion-xl-v1":{"mode":"image_generation","base_model":"stable-diffusion-xl-v1","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"256-x-256/dall-e-2":{"mode":"image_generation","base_model":"dall-e-2","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"512-x-512/50-steps/stability.stable-diffusion-xl-v0":{"mode":"image_generation","base_model":"stable-diffusion-xl-v0","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"512-x-512/dall-e-2":{"mode":"image_generation","base_model":"dall-e-2","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"512-x-512/max-steps/stability.stable-diffusion-xl-v0":{"mode":"image_generation","base_model":"stable-diffusion-xl-v0","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ai21.j2-mid-v1":{"mode":"chat","base_model":"j2-mid-v1","max_input_tokens":8191,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ai21.j2-ultra-v1":{"mode":"chat","base_model":"j2-ultra-v1","max_input_tokens":8191,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ai21.jamba-1-5-large-v1:0":{"mode":"chat","base_model":"jamba-1-5-large","deprecation_date":"2026-11-26","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_response_schema":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1"],"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ai21.jamba-1-5-mini-v1:0":{"mode":"chat","base_model":"jamba-1-5-mini","deprecation_date":"2026-11-26","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1"],"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ai21.jamba-instruct-v1:0":{"mode":"chat","base_model":"jamba-instruct","max_input_tokens":70000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","supports_system_messages":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/dall-e-2":{"mode":"image_generation","base_model":"dall-e-2","metadata":{"notes":"DALL-E 2 via AI/ML API - Reliable text-to-image generation"},"source":"https://docs.aimlapi.com/","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","metadata":{"notes":"DALL-E 3 via AI/ML API - High-quality text-to-image generation"},"source":"https://docs.aimlapi.com/","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/flux-pro":{"mode":"image_generation","base_model":"flux-pro","metadata":{"notes":"Flux Dev - Development version optimized for experimentation"},"source":"https://docs.aimlapi.com/","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/flux-pro/v1.1":{"mode":"image_generation","base_model":"flux-pro/v1.1","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/flux-pro/v1.1-ultra":{"mode":"image_generation","base_model":"flux-pro/v1.1-ultra","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/flux-realism":{"mode":"image_generation","base_model":"flux-realism","metadata":{"notes":"Flux Pro - Professional-grade image generation model"},"source":"https://docs.aimlapi.com/","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/flux/dev":{"mode":"image_generation","base_model":"flux/dev","metadata":{"notes":"Flux Dev - Development version optimized for experimentation"},"source":"https://docs.aimlapi.com/","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/flux/kontext-max/text-to-image":{"mode":"image_generation","base_model":"flux/kontext-max/text-to-image","metadata":{"notes":"Flux Pro v1.1 - Enhanced version with improved capabilities and 6x faster inference speed"},"source":"https://docs.aimlapi.com/","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/flux/kontext-pro/text-to-image":{"mode":"image_generation","base_model":"flux/kontext-pro/text-to-image","metadata":{"notes":"Flux Pro v1.1 - Enhanced version with improved capabilities and 6x faster inference speed"},"source":"https://docs.aimlapi.com/","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/flux/schnell":{"mode":"image_generation","base_model":"flux/schnell","metadata":{"notes":"Flux Schnell - Fast generation model optimized for speed"},"source":"https://docs.aimlapi.com/","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/google/imagen-4.0-ultra-generate-001":{"mode":"image_generation","base_model":"imagen-4.0-ultra-generate","metadata":{"notes":"Imagen 4.0 Ultra Generate API - Photorealistic image generation with precise text rendering"},"source":"https://docs.aimlapi.com/api-references/image-models/google/imagen-4-ultra-generate","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aiml/google/nano-banana-pro":{"mode":"image_generation","base_model":"nano-banana-pro","metadata":{"notes":"Gemini 3 Pro Image (Nano Banana Pro) - Advanced text-to-image generation with reasoning and 4K resolution support"},"source":"https://docs.aimlapi.com/api-references/image-models/google/gemini-3-pro-image-preview","supported_endpoints":["/v1/images/generations"],"provider":"aiml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon.nova-canvas-v1:0":{"mode":"image_generation","base_model":"nova-canvas","max_input_tokens":2600,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2600}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.writer.palmyra-x4-v1:0":{"mode":"chat","base_model":"palmyra-x4","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_pdf_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.writer.palmyra-x5-v1:0":{"mode":"chat","base_model":"palmyra-x5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_pdf_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"writer.palmyra-x4-v1:0":{"mode":"chat","base_model":"palmyra-x4","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_pdf_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"writer.palmyra-x5-v1:0":{"mode":"chat","base_model":"palmyra-x5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_pdf_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon.nova-lite-v1:0":{"mode":"chat","base_model":"nova-lite","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":10000,"range":{"min":1,"max":10000}}]},"amazon.nova-2-lite-v1:0":{"mode":"chat","base_model":"nova-2-lite","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon.nova-2-pro-preview-20251202-v1:0":{"mode":"chat","base_model":"nova-2-pro","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.amazon.nova-2-lite-v1:0":{"mode":"chat","base_model":"nova-2-lite","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.amazon.nova-2-pro-preview-20251202-v1:0":{"mode":"chat","base_model":"nova-2-pro","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.amazon.nova-2-lite-v1:0":{"mode":"chat","base_model":"nova-2-lite","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.amazon.nova-2-pro-preview-20251202-v1:0":{"mode":"chat","base_model":"nova-2-pro","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.amazon.nova-2-lite-v1:0":{"mode":"chat","base_model":"nova-2-lite","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.amazon.nova-2-pro-preview-20251202-v1:0":{"mode":"chat","base_model":"nova-2-pro","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon.nova-2-multimodal-embeddings-v1:0":{"mode":"embedding","base_model":"nova-2-multimodal-embeddings","max_input_tokens":8172,"max_tokens":8172,"output_vector_size":3072,"source":"https://us-east-1.console.aws.amazon.com/bedrock/home?region=us-east-1#/model-catalog/serverless/amazon.nova-2-multimodal-embeddings-v1:0","supports_embedding_image_input":true,"supports_image_input":true,"supports_video_input":true,"supports_audio_input":true,"provider":"bedrock"},"amazon.nova-micro-v1:0":{"mode":"chat","base_model":"nova-micro","max_input_tokens":128000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":10000,"range":{"min":1,"max":10000}}]},"amazon.nova-pro-v1:0":{"mode":"chat","base_model":"nova-pro","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":10000,"range":{"min":1,"max":10000}}]},"amazon.rerank-v1:0":{"mode":"rerank","base_model":"rerank","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"provider":"bedrock","max_document_chunks_per_query":100,"max_query_tokens":32000,"max_tokens_per_document_chunk":512,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon.titan-embed-image-v1":{"mode":"embedding","base_model":"titan-embed-image-v1","max_input_tokens":128,"max_tokens":128,"metadata":{"notes":"'supports_image_input' is a deprecated field. Use 'supports_embedding_image_input' instead."},"output_vector_size":1024,"source":"https://us-east-1.console.aws.amazon.com/bedrock/home?region=us-east-1#/providers?model=amazon.titan-image-generator-v1","supports_embedding_image_input":true,"supports_image_input":true,"provider":"bedrock"},"amazon.titan-embed-text-v1":{"mode":"embedding","base_model":"titan-embed-text-v1","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":1536,"provider":"bedrock"},"amazon.titan-embed-text-v2:0":{"mode":"embedding","base_model":"titan-embed-text","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":1024,"provider_specific_entry":{"bedrock_invocation_schema":"titan_v2"},"provider":"bedrock"},"amazon.titan-image-generator-v1":{"mode":"image_generation","base_model":"titan-image-generator-v1","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon.titan-image-generator-v2":{"mode":"image_generation","base_model":"titan-image-generator-v2","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon.titan-image-generator-v2:0":{"mode":"image_generation","base_model":"titan-image-generator","provider":"bedrock","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"twelvelabs.marengo-embed-2-7-v1:0":{"mode":"embedding","base_model":"marengo-embed-2-7","deprecation_date":"2026-11-30","max_input_tokens":77,"max_tokens":77,"output_vector_size":1024,"supports_embedding_image_input":true,"supports_image_input":true,"provider":"bedrock"},"us.twelvelabs.marengo-embed-2-7-v1:0":{"mode":"embedding","base_model":"marengo-embed-2-7","deprecation_date":"2026-11-30","max_input_tokens":77,"max_tokens":77,"output_vector_size":1024,"supports_embedding_image_input":true,"supports_image_input":true,"provider":"bedrock"},"eu.twelvelabs.marengo-embed-2-7-v1:0":{"mode":"embedding","base_model":"marengo-embed-2-7","deprecation_date":"2026-11-30","max_input_tokens":77,"max_tokens":77,"output_vector_size":1024,"supports_embedding_image_input":true,"supports_image_input":true,"provider":"bedrock"},"twelvelabs.pegasus-1-2-v1:0":{"mode":"chat","base_model":"pegasus-1-2","supports_video_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.twelvelabs.pegasus-1-2-v1:0":{"mode":"chat","base_model":"pegasus-1-2","supports_video_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.twelvelabs.pegasus-1-2-v1:0":{"mode":"chat","base_model":"pegasus-1-2","supports_video_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon.titan-text-express-v1":{"mode":"chat","base_model":"titan-text-express-v1","max_input_tokens":42000,"max_output_tokens":8000,"max_tokens":8000,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8000,"range":{"min":1,"max":8000}}]},"amazon.titan-text-lite-v1":{"mode":"chat","base_model":"titan-text-lite-v1","max_input_tokens":42000,"max_output_tokens":4000,"max_tokens":4000,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"amazon.titan-text-premier-v1:0":{"mode":"chat","base_model":"titan-text-premier","max_input_tokens":42000,"max_output_tokens":32000,"max_tokens":32000,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32000,"range":{"min":1,"max":32000}}]},"anthropic.claude-3-5-haiku-20241022-v1:0":{"mode":"chat","base_model":"claude-3-5-haiku","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"prompt_cache_min_tokens":2048,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2"],"deprecation_date":"2026-06-19","is_deprecated":true,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","supports_web_search":true,"tool_use_system_prompt_tokens":346,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"anthropic.claude-haiku-4-5@20251001":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_streaming":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","tool_use_system_prompt_tokens":346,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_web_search":true},"anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":1000000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anthropic.claude-3-5-sonnet-20241022-v2:0":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anthropic.claude-3-7-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-7-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anthropic.claude-3-7-sonnet-20250219-v1:0":{"mode":"chat","base_model":"claude-3-7-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":8192}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","deprecation_date":"2026-09-10","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","ca-central-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"anthropic.claude-3-opus-20240229-v1:0":{"mode":"chat","base_model":"claude-3-opus","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anthropic.claude-3-sonnet-20240229-v1:0":{"mode":"chat","base_model":"claude-3-sonnet","deprecation_date":"2026-07-30","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anthropic.claude-opus-4-1-20250805-v1:0":{"mode":"chat","base_model":"claude-opus-4-1","deprecation_date":"2027-01-08","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2"],"is_deprecated":true,"source":"https://aws.amazon.com/bedrock/pricing/","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anthropic.claude-opus-4-20250514-v1:0":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"anthropic.claude-opus-4-5-20251101-v1:0":{"mode":"chat","base_model":"claude-opus-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"high","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1},"disabledCondition":{"paramId":"thinking","operator":"eq","value":true,"setValue":1,"disabledText":"Temperature is disabled when extended thinking is enabled."}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"effort","label":"Effort","helpText":"Set the effort level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"thinking","label":"Extended thinking","helpText":"Enable the extended thinking parameter","type":"boolean","default":true},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"anthropic.claude-opus-4-6-v1":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1},"disabledCondition":{"paramId":"thinking","operator":"eq","value":true,"setValue":1,"disabledText":"Temperature is disabled when extended thinking is enabled."}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"effort","label":"Effort","helpText":"Set the effort level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"thinking","label":"Extended thinking","helpText":"Enable the extended thinking parameter","type":"boolean","default":true},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"global.anthropic.claude-opus-4-6-v1":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_adaptive_thinking":true,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"us.anthropic.claude-opus-4-6-v1":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_adaptive_thinking":true,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"eu.anthropic.claude-opus-4-6-v1":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_adaptive_thinking":true,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"au.anthropic.claude-opus-4-6-v1":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_adaptive_thinking":true,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"anthropic.claude-sonnet-4-20250514-v1:0":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-10-14","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"bedrock_converse_supports_strict_tools":false,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_assistant_prefill":true,"supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"supports_assistant_prefill":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anyscale/HuggingFaceH4/zephyr-7b-beta":{"mode":"chat","base_model":"zephyr-7b","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/codellama/CodeLlama-34b-Instruct-hf":{"mode":"chat","base_model":"codellama-34b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/codellama/CodeLlama-70b-Instruct-hf":{"mode":"chat","base_model":"codellama-70b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://docs.anyscale.com/preview/endpoints/text-generation/supported-models/codellama-CodeLlama-70b-Instruct-hf","provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/google/gemma-7b-it":{"mode":"chat","base_model":"gemma-7b-it","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://docs.anyscale.com/preview/endpoints/text-generation/supported-models/google-gemma-7b-it","provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anyscale/meta-llama/Llama-2-13b-chat-hf":{"mode":"chat","base_model":"llama-2-13b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/meta-llama/Llama-2-70b-chat-hf":{"mode":"chat","base_model":"llama-2-70b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/meta-llama/Llama-2-7b-chat-hf":{"mode":"chat","base_model":"llama-2-7b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/meta-llama/Meta-Llama-3-70B-Instruct":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://docs.anyscale.com/preview/endpoints/text-generation/supported-models/meta-llama-Meta-Llama-3-70B-Instruct","provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/meta-llama/Meta-Llama-3-8B-Instruct":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://docs.anyscale.com/preview/endpoints/text-generation/supported-models/meta-llama-Meta-Llama-3-8B-Instruct","provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"anyscale/mistralai/Mistral-7B-Instruct-v0.1":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://docs.anyscale.com/preview/endpoints/text-generation/supported-models/mistralai-Mistral-7B-Instruct-v0.1","supports_function_calling":true,"provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/mistralai/Mixtral-8x22B-Instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x22b-instruct","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"source":"https://docs.anyscale.com/preview/endpoints/text-generation/supported-models/mistralai-Mixtral-8x22B-Instruct-v0.1","supports_function_calling":true,"provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":65536}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"anyscale/mistralai/Mixtral-8x7B-Instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://docs.anyscale.com/preview/endpoints/text-generation/supported-models/mistralai-Mixtral-8x7B-Instruct-v0.1","supports_function_calling":true,"provider":"anyscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.amazon.nova-lite-v1:0":{"mode":"chat","base_model":"nova-lite","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.amazon.nova-micro-v1:0":{"mode":"chat","base_model":"nova-micro","max_input_tokens":128000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.amazon.nova-pro-v1:0":{"mode":"chat","base_model":"nova-pro","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","deprecation_date":"2026-07-30","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.anthropic.claude-3-5-sonnet-20241022-v2:0":{"mode":"chat","base_model":"claude-3-5-sonnet","deprecation_date":"2026-07-30","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_cache_point":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","deprecation_date":"2026-09-10","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","tool_use_system_prompt_tokens":346,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.anthropic.claude-3-sonnet-20240229-v1:0":{"mode":"chat","base_model":"claude-3-sonnet","deprecation_date":"2026-07-30","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.anthropic.claude-sonnet-4-20250514-v1:0":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-10-14","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"bedrock_converse_supports_strict_tools":false,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"assemblyai/best":{"mode":"audio_transcription","base_model":"best","provider":"assemblyai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"assemblyai/nano":{"mode":"audio_transcription","base_model":"nano","provider":"assemblyai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"au.anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"azure/ada":{"mode":"embedding","base_model":"ada","max_input_tokens":8191,"max_tokens":8191,"provider":"azure"},"azure/codex-mini":{"mode":"responses","base_model":"codex-mini","deprecation_date":"2026-11-15","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/command-r-plus":{"mode":"chat","base_model":"command-r-plus","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","deprecation_date":"2026-10-19","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/claude-opus-4-5":{"mode":"chat","base_model":"claude-opus-4-5","deprecation_date":"2026-10-19","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"provider":"azure","supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/claude-opus-4-6":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/claude-opus-4-1":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/claude-sonnet-4-5":{"mode":"chat","base_model":"claude-sonnet-4-5","deprecation_date":"2026-10-19","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/computer-use-preview":{"mode":"chat","base_model":"computer-use","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/container":{"mode":"chat","base_model":"container","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/model_router":{"mode":"chat","base_model":"model-router","deprecation_date":"2027-05-20","max_input_tokens":200000,"max_output_tokens":32768,"max_tokens":32768,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-services/","comment":"Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/eu/gpt-4o-2024-08-06":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-02-27","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/eu/gpt-4o-2024-11-20":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-03-01","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/eu/gpt-4o-mini-2024-07-18":{"mode":"chat","base_model":"gpt-4o-mini","deprecation_date":"2027-04-14","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presencePenalty","label":"Presence penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/eu/gpt-4o-mini-realtime-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-mini-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/eu/gpt-4o-realtime-preview-2024-10-01":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/eu/gpt-4o-realtime-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/eu/gpt-5-2025-08-07":{"mode":"chat","base_model":"gpt-5","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5-mini-2025-08-07":{"mode":"chat","base_model":"gpt-5-mini","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.1-chat":{"mode":"chat","base_model":"gpt-5.1-chat","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/eu/gpt-5.1-codex":{"mode":"responses","base_model":"gpt-5.1-codex","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/eu/gpt-5.1-codex-mini":{"mode":"responses","base_model":"gpt-5.1-codex-mini","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/eu/gpt-5-nano-2025-08-07":{"mode":"chat","base_model":"gpt-5-nano","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/eu/o1-2024-12-17":{"mode":"chat","base_model":"o1","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/eu/o1-mini-2024-09-12":{"mode":"chat","base_model":"o1-mini","max_input_tokens":128000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":65536}}]},"azure/eu/o1-preview-2024-09-12":{"mode":"chat","base_model":"o1","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/eu/o3-mini-2025-01-31":{"mode":"chat","base_model":"o3-mini","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/global-standard/gpt-4o-2024-08-06":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-02-27","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/global-standard/gpt-4o-2024-11-20":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-03-01","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/global-standard/gpt-4o-mini":{"mode":"chat","base_model":"gpt-4o-mini","deprecation_date":"2027-04-14","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presencePenalty","label":"Presence penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/global/gpt-4o-2024-08-06":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-02-27","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/global/gpt-4o-2024-11-20":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-03-01","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/global/gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/global/gpt-5.1-chat":{"mode":"chat","base_model":"gpt-5.1-chat","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/global/gpt-5.1-codex":{"mode":"responses","base_model":"gpt-5.1-codex","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/global/gpt-5.1-codex-mini":{"mode":"responses","base_model":"gpt-5.1-codex-mini","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-3.5-turbo":{"mode":"chat","base_model":"gpt-3.5-turbo","max_input_tokens":4097,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-3.5-turbo-0125":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2025-03-31","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"is_deprecated":true},"azure/gpt-3.5-turbo-instruct-0914":{"mode":"completion","base_model":"gpt-3.5-turbo-instruct","max_input_tokens":4097,"max_tokens":4097,"provider":"azure_text","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4097}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-35-turbo":{"mode":"chat","base_model":"gpt-3.5-turbo","max_input_tokens":4097,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-35-turbo-0125":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2025-05-31","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"is_deprecated":true},"azure/gpt-35-turbo-0301":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2025-02-13","max_input_tokens":4097,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"is_deprecated":true},"azure/gpt-35-turbo-0613":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2025-02-13","max_input_tokens":4097,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"is_deprecated":true},"azure/gpt-35-turbo-1106":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2025-03-31","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"is_deprecated":true},"azure/gpt-35-turbo-16k":{"mode":"chat","base_model":"gpt-3.5-turbo-16k","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-35-turbo-16k-0613":{"mode":"chat","base_model":"gpt-3.5-turbo-16k","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-35-turbo-instruct":{"mode":"completion","base_model":"gpt-3.5-turbo-instruct","max_input_tokens":4097,"max_tokens":4097,"provider":"azure_text","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4097}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-35-turbo-instruct-0914":{"mode":"completion","base_model":"gpt-3.5-turbo-instruct","max_input_tokens":4097,"max_tokens":4097,"provider":"azure_text","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4097}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4":{"mode":"chat","base_model":"gpt-4","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-4-0125-preview":{"mode":"chat","base_model":"gpt-4","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-4-0613":{"mode":"chat","base_model":"gpt-4","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-4-1106-preview":{"mode":"chat","base_model":"gpt-4","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-4-32k":{"mode":"chat","base_model":"gpt-4-32k","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-4-32k-0613":{"mode":"chat","base_model":"gpt-4-32k","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-4-turbo":{"mode":"chat","base_model":"gpt-4-turbo","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4-turbo-2024-04-09":{"mode":"chat","base_model":"gpt-4-turbo","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4-turbo-vision-preview":{"mode":"chat","base_model":"gpt-4-turbo-vision","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","deprecation_date":"2027-04-14","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/gpt-4.1-2025-04-14":{"mode":"chat","base_model":"gpt-4.1","deprecation_date":"2026-11-04","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/gpt-4.1-mini":{"mode":"chat","base_model":"gpt-4.1-mini","deprecation_date":"2027-04-14","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/gpt-4.1-mini-2025-04-14":{"mode":"chat","base_model":"gpt-4.1-mini","deprecation_date":"2026-11-04","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/gpt-4.1-nano":{"mode":"chat","base_model":"gpt-4.1-nano","deprecation_date":"2027-04-14","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/gpt-4.1-nano-2025-04-14":{"mode":"chat","base_model":"gpt-4.1-nano","deprecation_date":"2026-11-04","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/gpt-4.5-preview":{"mode":"chat","base_model":"gpt-4.5","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/gpt-4o":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2027-04-14","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_web_search":true},"azure/gpt-4o-2024-05-13":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-10-01","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/gpt-4o-2024-08-06":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-02-27","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/gpt-4o-2024-11-20":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-03-01","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/gpt-audio-2025-08-28":{"mode":"chat","base_model":"gpt-audio","deprecation_date":"2027-03-02","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-audio-mini-2025-10-06":{"mode":"chat","base_model":"gpt-audio-mini","deprecation_date":"2027-04-06","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4o-audio-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-audio","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/gpt-4o-mini":{"mode":"chat","base_model":"gpt-4o-mini","deprecation_date":"2027-04-14","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presencePenalty","label":"Presence penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/gpt-4o-mini-2024-07-18":{"mode":"chat","base_model":"gpt-4o-mini","deprecation_date":"2027-04-14","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presencePenalty","label":"Presence penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/gpt-4o-mini-audio-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-mini-audio","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4o-mini-realtime-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-mini-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-realtime-2025-08-28":{"mode":"chat","base_model":"gpt-realtime","deprecation_date":"2027-03-02","max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-realtime-mini-2025-10-06":{"mode":"chat","base_model":"gpt-realtime-mini","max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4o-mini-transcribe":{"mode":"audio_transcription","base_model":"gpt-4o-mini-transcribe","max_input_tokens":16000,"max_output_tokens":2000,"supported_endpoints":["/v1/audio/transcriptions"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4o-mini-tts":{"mode":"audio_speech","base_model":"gpt-4o-mini-tts","supported_endpoints":["/v1/audio/speech"],"supported_modalities":["text","audio"],"supported_output_modalities":["audio"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4o-realtime-preview-2024-10-01":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4o-realtime-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4o-transcribe":{"mode":"audio_transcription","base_model":"gpt-4o-transcribe","deprecation_date":"2026-10-15","max_input_tokens":16000,"max_output_tokens":2000,"supported_endpoints":["/v1/audio/transcriptions"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-4o-transcribe-diarize":{"mode":"audio_transcription","base_model":"gpt-4o-transcribe","deprecation_date":"2027-04-15","max_input_tokens":16000,"max_output_tokens":2000,"supported_endpoints":["/v1/audio/transcriptions"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.1-2025-11-13":{"mode":"chat","base_model":"gpt-5.1","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_minimal_reasoning_effort":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/gpt-5.1-chat-2025-11-13":{"mode":"chat","base_model":"gpt-5.1-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_native_streaming":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.1-codex-2025-11-13":{"mode":"responses","base_model":"gpt-5.1-codex","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.1-codex-mini-2025-11-13":{"mode":"responses","base_model":"gpt-5.1-codex-mini","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5":{"mode":"chat","base_model":"gpt-5","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/gpt-5-2025-08-07":{"mode":"chat","base_model":"gpt-5","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/gpt-5-chat":{"mode":"chat","base_model":"gpt-5-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://azure.microsoft.com/en-us/blog/gpt-5-in-azure-ai-foundry-the-future-of-ai-apps-and-agents-starts-here/","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5-chat-latest":{"mode":"chat","base_model":"gpt-5-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/gpt-5-codex":{"mode":"responses","base_model":"gpt-5-codex","deprecation_date":"2027-03-17","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/gpt-5-mini-2025-08-07":{"mode":"chat","base_model":"gpt-5-mini","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/gpt-5-nano-2025-08-07":{"mode":"chat","base_model":"gpt-5-nano","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/gpt-5-pro":{"mode":"responses","base_model":"gpt-5-pro","deprecation_date":"2027-04-07","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://learn.microsoft.com/en-us/azure/ai-foundry/foundry-models/concepts/models-sold-directly-by-azure?pivots=azure-openai&tabs=global-standard-aoai%2Cstandard-chat-completions%2Cglobal-standard#gpt-5","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/gpt-5.1-chat":{"mode":"chat","base_model":"gpt-5.1-chat","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.1-codex":{"mode":"responses","base_model":"gpt-5.1-codex","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.1-codex-max":{"mode":"responses","base_model":"gpt-5.1-codex-max","deprecation_date":"2027-05-18","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.1-codex-mini":{"mode":"responses","base_model":"gpt-5.1-codex-mini","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.2":{"mode":"chat","base_model":"gpt-5.2","deprecation_date":"2027-06-08","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.2-2025-12-11":{"mode":"chat","base_model":"gpt-5.2","deprecation_date":"2027-06-08","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.2-chat":{"mode":"chat","base_model":"gpt-5.2-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.2-chat-2025-12-11":{"mode":"chat","base_model":"gpt-5.2-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.2-codex":{"mode":"responses","base_model":"gpt-5.2-codex","deprecation_date":"2027-07-13","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.2-pro":{"mode":"responses","base_model":"gpt-5.2-pro","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-5.2-pro-2025-12-11":{"mode":"responses","base_model":"gpt-5.2-pro","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","deprecation_date":"2026-10-23","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/hd/1024-x-1024/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/hd/1024-x-1792/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/hd/1792-x-1024/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/high/1024-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/high/1024-x-1536/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/high/1536-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/low/1024-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/low/1024-x-1536/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/low/1536-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/medium/1024-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/medium/1024-x-1536/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/medium/1536-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","deprecation_date":"2027-04-07","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","deprecation_date":"2026-12-16","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","deprecation_date":"2026-12-16","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/low/1024-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/low/1024-x-1536/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/low/1536-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/medium/1024-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/medium/1024-x-1536/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/medium/1536-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/high/1024-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/high/1024-x-1536/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/high/1536-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-large-2402":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-large-latest":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azuremarketplace.microsoft.com/en/marketplace/apps/000-000.mistral-ai-large-2407-offer?tab=Overview","supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/o1":{"mode":"chat","base_model":"o1","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/o1-2024-12-17":{"mode":"chat","base_model":"o1","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/o1-mini":{"mode":"chat","base_model":"o1-mini","max_input_tokens":128000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":65536}}]},"azure/o1-mini-2024-09-12":{"mode":"chat","base_model":"o1-mini","max_input_tokens":128000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":65536}}]},"azure/o1-preview":{"mode":"chat","base_model":"o1","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/o1-preview-2024-09-12":{"mode":"chat","base_model":"o1","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/o3":{"mode":"chat","base_model":"o3","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/o3-2025-04-16":{"mode":"chat","base_model":"o3","deprecation_date":"2026-04-16","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/o3-deep-research":{"mode":"responses","base_model":"o3","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/o3-mini":{"mode":"chat","base_model":"o3-mini","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/o3-mini-2025-01-31":{"mode":"chat","base_model":"o3-mini","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/o3-pro":{"mode":"responses","base_model":"o3-pro","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/o3-pro-2025-06-10":{"mode":"responses","base_model":"o3-pro","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/o4-mini":{"mode":"chat","base_model":"o4-mini","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/o4-mini-2025-04-16":{"mode":"chat","base_model":"o4-mini","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/standard/1024-x-1024/dall-e-2":{"mode":"image_generation","base_model":"dall-e-2","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/standard/1024-x-1024/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/standard/1024-x-1792/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/standard/1792-x-1024/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/text-embedding-3-large":{"mode":"embedding","base_model":"text-embedding-3-large","deprecation_date":"2028-02-09","max_input_tokens":8191,"max_tokens":8191,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure"},"azure/text-embedding-3-small":{"mode":"embedding","base_model":"text-embedding-3-small","deprecation_date":"2026-04-30","max_input_tokens":8191,"max_tokens":8191,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","is_deprecated":true},"azure/text-embedding-ada-002":{"mode":"embedding","base_model":"text-embedding-ada-002","deprecation_date":"2028-02-09","max_input_tokens":8191,"max_tokens":8191,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure"},"azure/speech/azure-tts":{"mode":"audio_speech","base_model":"speech/azure-tts","source":"https://azure.microsoft.com/en-us/pricing/calculator/","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/speech/azure-tts-hd":{"mode":"audio_speech","base_model":"speech/azure-tts-hd","source":"https://azure.microsoft.com/en-us/pricing/calculator/","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/tts-1":{"mode":"audio_speech","base_model":"tts-1","deprecation_date":"2026-12-15","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/tts-1-hd":{"mode":"audio_speech","base_model":"tts-1-hd","deprecation_date":"2026-12-15","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/us/gpt-4.1-2025-04-14":{"mode":"chat","base_model":"gpt-4.1","deprecation_date":"2026-11-04","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/us/gpt-4.1-mini-2025-04-14":{"mode":"chat","base_model":"gpt-4.1-mini","deprecation_date":"2026-11-04","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/us/gpt-4.1-nano-2025-04-14":{"mode":"chat","base_model":"gpt-4.1-nano","deprecation_date":"2026-11-04","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/us/gpt-4o-2024-08-06":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-02-27","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/us/gpt-4o-2024-11-20":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-03-01","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/us/gpt-4o-mini-2024-07-18":{"mode":"chat","base_model":"gpt-4o-mini","deprecation_date":"2027-04-14","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presencePenalty","label":"Presence penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/us/gpt-4o-mini-realtime-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-mini-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/us/gpt-4o-realtime-preview-2024-10-01":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/us/gpt-4o-realtime-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/us/gpt-5-2025-08-07":{"mode":"chat","base_model":"gpt-5","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5-mini-2025-08-07":{"mode":"chat","base_model":"gpt-5-mini","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5-nano-2025-08-07":{"mode":"chat","base_model":"gpt-5-nano","deprecation_date":"2027-02-09","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.1-chat":{"mode":"chat","base_model":"gpt-5.1-chat","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/us/gpt-5.1-codex":{"mode":"responses","base_model":"gpt-5.1-codex","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/us/gpt-5.1-codex-mini":{"mode":"responses","base_model":"gpt-5.1-codex-mini","deprecation_date":"2027-05-15","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/us/o1-2024-12-17":{"mode":"chat","base_model":"o1","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/us/o1-mini-2024-09-12":{"mode":"chat","base_model":"o1-mini","max_input_tokens":128000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":65536}}]},"azure/us/o1-preview-2024-09-12":{"mode":"chat","base_model":"o1","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/us/o3-2025-04-16":{"mode":"chat","base_model":"o3","deprecation_date":"2026-04-16","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","is_deprecated":true,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/us/o3-mini-2025-01-31":{"mode":"chat","base_model":"o3-mini","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/us/o4-mini-2025-04-16":{"mode":"chat","base_model":"o4-mini","deprecation_date":"2026-11-19","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"azure/whisper-1":{"mode":"audio_transcription","base_model":"whisper-1","deprecation_date":"2026-12-15","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Cohere-embed-v3-english":{"mode":"embedding","base_model":"embed-v3-english","max_input_tokens":512,"max_tokens":512,"output_vector_size":1024,"source":"https://azuremarketplace.microsoft.com/en-us/marketplace/apps/cohere.cohere-embed-v3-english-offer?tab=PlansAndPrice","supports_embedding_image_input":true,"provider":"azure"},"azure/Cohere-embed-v3-multilingual":{"mode":"embedding","base_model":"embed-v3-multilingual","max_input_tokens":512,"max_tokens":512,"output_vector_size":1024,"source":"https://azuremarketplace.microsoft.com/en-us/marketplace/apps/cohere.cohere-embed-v3-english-offer?tab=PlansAndPrice","supports_embedding_image_input":true,"provider":"azure"},"azure/FLUX-1.1-pro":{"mode":"image_generation","base_model":"flux-1.1-pro","source":"https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/black-forest-labs-flux-1-kontext-pro-and-flux1-1-pro-now-available-in-azure-ai-f/4434659","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/FLUX.1-Kontext-pro":{"mode":"image_generation","base_model":"flux.1-kontext-pro","source":"https://azuremarketplace.microsoft.com/pt-br/marketplace/apps/cohere.cohere-embed-4-offer?tab=PlansAndPrice","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/flux.2-pro":{"mode":"image_generation","base_model":"flux.2-pro","source":"https://ai.azure.com/explore/models/flux.2-pro/version/1/registry/azureml-blackforestlabs","supported_endpoints":["/v1/images/generations"],"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Llama-3.2-11B-Vision-Instruct":{"mode":"chat","base_model":"llama-3.2-11b-vision-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"source":"https://azuremarketplace.microsoft.com/en/marketplace/apps/metagenai.meta-llama-3-2-11b-vision-instruct-offer?tab=Overview","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Llama-3.2-90B-Vision-Instruct":{"mode":"chat","base_model":"llama-3.2-90b-vision-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"source":"https://azuremarketplace.microsoft.com/en/marketplace/apps/metagenai.meta-llama-3-2-90b-vision-instruct-offer?tab=Overview","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"source":"https://azuremarketplace.microsoft.com/en/marketplace/apps/metagenai.llama-3-3-70b-instruct-offer?tab=Overview","supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Llama-4-Maverick-17B-128E-Instruct-FP8":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct-fp8","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://azure.microsoft.com/en-us/blog/introducing-the-llama-4-herd-in-azure-ai-foundry-and-azure-databricks/","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_input_tokens":10000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://azure.microsoft.com/en-us/blog/introducing-the-llama-4-herd-in-azure-ai-foundry-and-azure-databricks/","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/Meta-Llama-3-70B-Instruct":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":2048,"max_tokens":2048,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":2048}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/Meta-Llama-3.1-405B-Instruct":{"mode":"chat","base_model":"llama-3.1-405b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"source":"https://azuremarketplace.microsoft.com/en-us/marketplace/apps/metagenai.meta-llama-3-1-405b-instruct-offer?tab=PlansAndPrice","supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Meta-Llama-3.1-70B-Instruct":{"mode":"chat","base_model":"llama-3.1-70b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"source":"https://azuremarketplace.microsoft.com/en-us/marketplace/apps/metagenai.meta-llama-3-1-70b-instruct-offer?tab=PlansAndPrice","supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Meta-Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"source":"https://azuremarketplace.microsoft.com/en-us/marketplace/apps/metagenai.meta-llama-3-1-8b-instruct-offer?tab=PlansAndPrice","supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3-medium-128k-instruct":{"mode":"chat","base_model":"phi-3-medium-128k-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3-medium-4k-instruct":{"mode":"chat","base_model":"phi-3-medium-4k-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3-mini-128k-instruct":{"mode":"chat","base_model":"phi-3-mini-128k-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3-mini-4k-instruct":{"mode":"chat","base_model":"phi-3-mini-4k-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3-small-128k-instruct":{"mode":"chat","base_model":"phi-3-small-128k-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3-small-8k-instruct":{"mode":"chat","base_model":"phi-3-small-8k-instruct","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3.5-MoE-instruct":{"mode":"chat","base_model":"phi-3.5-moe-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3.5-mini-instruct":{"mode":"chat","base_model":"phi-3.5-mini-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-3.5-vision-instruct":{"mode":"chat","base_model":"phi-3.5-vision-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/phi-3/","supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-4":{"mode":"chat","base_model":"phi-4","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://techcommunity.microsoft.com/blog/machinelearningblog/affordable-innovation-unveiling-the-pricing-of-phi-3-slms-on-models-as-a-service/4156495","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presencePenalty","label":"Presence penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"azure/Phi-4-mini-instruct":{"mode":"chat","base_model":"phi-4-mini-instruct","max_input_tokens":131072,"max_output_tokens":4096,"max_tokens":4096,"source":"https://techcommunity.microsoft.com/blog/Azure-AI-Services-blog/announcing-new-phi-pricing-empowering-your-business-with-small-language-models/4395112","supports_function_calling":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-4-multimodal-instruct":{"mode":"chat","base_model":"phi-4-multimodal-instruct","max_input_tokens":131072,"max_output_tokens":4096,"max_tokens":4096,"source":"https://techcommunity.microsoft.com/blog/Azure-AI-Services-blog/announcing-new-phi-pricing-empowering-your-business-with-small-language-models/4395112","supports_audio_input":true,"supports_function_calling":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-4-mini-reasoning":{"mode":"chat","base_model":"phi-4-mini","max_input_tokens":131072,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/microsoft/","supports_function_calling":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/Phi-4-reasoning":{"mode":"chat","base_model":"phi-4","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/microsoft/","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presencePenalty","label":"Presence penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/mistral-document-ai-2505":{"mode":"ocr","base_model":"mistral-document-ai","supported_endpoints":["/v1/ocr"],"source":"https://devblogs.microsoft.com/foundry/whats-new-in-azure-ai-foundry-august-2025/#mistral-document-ai-(ocr)-%E2%80%94-serverless-in-foundry","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/doc-intelligence/prebuilt-read":{"mode":"ocr","base_model":"doc-intelligence/prebuilt-read","supported_endpoints":["/v1/ocr"],"source":"https://azure.microsoft.com/en-us/pricing/details/ai-document-intelligence/","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/doc-intelligence/prebuilt-layout":{"mode":"ocr","base_model":"doc-intelligence/prebuilt-layout","supported_endpoints":["/v1/ocr"],"source":"https://azure.microsoft.com/en-us/pricing/details/ai-document-intelligence/","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/doc-intelligence/prebuilt-document":{"mode":"ocr","base_model":"doc-intelligence/prebuilt-document","supported_endpoints":["/v1/ocr"],"source":"https://azure.microsoft.com/en-us/pricing/details/ai-document-intelligence/","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/MAI-DS-R1":{"mode":"chat","base_model":"mai-ds-r1","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/microsoft/","supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/cohere-rerank-v3-english":{"mode":"rerank","base_model":"rerank-v3-english","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"azure","max_query_tokens":2048,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/cohere-rerank-v3-multilingual":{"mode":"rerank","base_model":"rerank-v3-multilingual","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"azure","max_query_tokens":2048,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/cohere-rerank-v3.5":{"mode":"rerank","base_model":"rerank","max_input_tokens":4096,"max_output_tokens":4096,"max_query_tokens":2048,"max_tokens":4096,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/cohere-rerank-v4.0-pro":{"mode":"rerank","base_model":"rerank-v4.0-pro","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-cohere-rerank-4-0-in-microsoft-foundry/4477076","provider":"azure","max_query_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/cohere-rerank-v4.0-fast":{"mode":"rerank","base_model":"rerank-v4.0-fast","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-cohere-rerank-4-0-in-microsoft-foundry/4477076","provider":"azure","max_query_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/deepseek-v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/deepseek-v3.2-speciale":{"mode":"chat","base_model":"deepseek-v3.2","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/deepseek-r1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://techcommunity.microsoft.com/blog/machinelearningblog/deepseek-r1-improved-performance-higher-limits-and-transparent-pricing/4386367","supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}}]},"azure/deepseek-v3":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://techcommunity.microsoft.com/blog/machinelearningblog/announcing-deepseek-v3-on-azure-ai-foundry-and-github/4390438","supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}}]},"azure/deepseek-v3-0324":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://techcommunity.microsoft.com/blog/machinelearningblog/announcing-deepseek-v3-on-azure-ai-foundry-and-github/4390438","supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}}]},"azure/embed-v-4-0":{"mode":"embedding","base_model":"embed-v-4-0","max_input_tokens":128000,"max_tokens":128000,"output_vector_size":3072,"source":"https://azuremarketplace.microsoft.com/pt-br/marketplace/apps/cohere.cohere-embed-4-offer?tab=PlansAndPrice","supported_endpoints":["/v1/embeddings"],"supported_modalities":["text","image"],"supports_embedding_image_input":true,"provider":"azure"},"azure/global/grok-3":{"mode":"chat","base_model":"grok-3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://devblogs.microsoft.com/foundry/announcing-grok-3-and-grok-3-mini-on-azure-ai-foundry/","supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/global/grok-3-mini":{"mode":"chat","base_model":"grok-3-mini","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://devblogs.microsoft.com/foundry/announcing-grok-3-and-grok-3-mini-on-azure-ai-foundry/","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/grok-3":{"mode":"chat","base_model":"grok-3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/","supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/grok-3-mini":{"mode":"chat","base_model":"grok-3-mini","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/grok-4":{"mode":"chat","base_model":"grok-4","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/grok-4-fast-non-reasoning":{"mode":"chat","base_model":"grok-4-fast-non","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/grok-4-fast-reasoning":{"mode":"chat","base_model":"grok-4-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/grok-code-fast-1":{"mode":"chat","base_model":"grok-code-fast-1","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/grok/","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/jais-30b-chat":{"mode":"chat","base_model":"jais-30b-chat","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://azure.microsoft.com/en-us/products/ai-services/ai-foundry/models/jais-30b-chat","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/jamba-instruct":{"mode":"chat","base_model":"jamba-instruct","max_input_tokens":70000,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","deprecation_date":"2027-01-26","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/kimi-k2-5-now-in-microsoft-foundry/4492321","supports_function_calling":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_prompt_caching":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/ministral-3b":{"mode":"chat","base_model":"ministral-3b","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azuremarketplace.microsoft.com/en/marketplace/apps/000-000.ministral-3b-2410-offer?tab=Overview","supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-large":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-large-2407":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azuremarketplace.microsoft.com/en/marketplace/apps/000-000.mistral-ai-large-2407-offer?tab=Overview","supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-large-3":{"mode":"chat","base_model":"mistral-large-3","max_input_tokens":256000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://azure.microsoft.com/en-us/blog/introducing-mistral-large-3-in-microsoft-foundry-open-capable-and-ready-for-production-workloads/","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-medium-2505":{"mode":"chat","base_model":"mistral-medium","max_input_tokens":131072,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-nemo":{"mode":"chat","base_model":"mistral-nemo","max_input_tokens":131072,"max_output_tokens":4096,"max_tokens":4096,"source":"https://azuremarketplace.microsoft.com/en/marketplace/apps/000-000.mistral-nemo-12b-2407?tab=PlansAndPrice","supports_function_calling":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-small":{"mode":"chat","base_model":"mistral-small","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/mistral-small-2503":{"mode":"chat","base_model":"mistral-small","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"babbage-002":{"mode":"completion","base_model":"babbage-002","deprecation_date":"2026-09-28","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","provider":"text-completion-openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":false,"supports_tool_choice":false,"supports_parallel_function_calling":false,"supports_reasoning":false,"supports_response_schema":false,"supports_prompt_caching":false,"supports_web_search":false,"supports_service_tier":false,"supports_assistant_prefill":false,"supports_system_messages":false},"bedrock/*/1-month-commitment/cohere.command-light-text-v14":{"mode":"chat","base_model":"command-light-text-v14","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/*/1-month-commitment/cohere.command-text-v14":{"mode":"chat","base_model":"command-text-v14","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/*/6-month-commitment/cohere.command-light-text-v14":{"mode":"chat","base_model":"command-light-text-v14","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/*/6-month-commitment/cohere.command-text-v14":{"mode":"chat","base_model":"command-text-v14","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/1-month-commitment/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/6-month-commitment/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/moonshotai.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-northeast-1/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/moonshotai.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-south-1/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/ap-south-1/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-south-1/deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-south-1/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-south-1/moonshotai.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-south-1/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-south-1/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-southeast-3/deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-southeast-3/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-southeast-3/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ap-southeast-3/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/ca-central-1/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/ca-central-1/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-north-1/deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-north-1/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-north-1/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/1-month-commitment/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/1-month-commitment/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/6-month-commitment/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/6-month-commitment/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-central-1/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-west-1/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/eu-west-1/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-west-1/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-west-1/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-west-2/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/eu-west-2/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-west-2/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-west-2/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-west-3/mistral.mistral-7b-instruct-v0:2":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8191}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/eu-west-3/mistral.mistral-large-2402-v1:0":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-west-3/mistral.mixtral-8x7b-instruct-v0:1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-south-1/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/eu-south-1/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/invoke/anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"metadata":{"notes":"Anthropic via Invoke route does not currently support pdf input."},"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/sa-east-1/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/sa-east-1/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/sa-east-1/deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/sa-east-1/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/sa-east-1/moonshotai.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/sa-east-1/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/sa-east-1/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/1-month-commitment/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/1-month-commitment/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/1-month-commitment/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/6-month-commitment/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/6-month-commitment/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/6-month-commitment/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/us-east-1/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/mistral.mistral-7b-instruct-v0:2":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8191}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/us-east-1/mistral.mistral-large-2402-v1:0":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/mistral.mixtral-8x7b-instruct-v0:1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/moonshotai.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-1/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-2/deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-2/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-2/moonshotai.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-2/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-east-2/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-east-1/amazon.nova-pro-v1:0":{"mode":"chat","base_model":"nova-pro","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-east-1/amazon.titan-embed-text-v1":{"mode":"embedding","base_model":"titan-embed-text-v1","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":1536,"provider":"bedrock"},"bedrock/us-gov-east-1/amazon.titan-embed-text-v2:0":{"mode":"embedding","base_model":"titan-embed-text","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":1024,"provider":"bedrock"},"bedrock/us-gov-east-1/amazon.titan-text-express-v1":{"mode":"chat","base_model":"titan-text-express-v1","max_input_tokens":42000,"max_output_tokens":8000,"max_tokens":8000,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-east-1/amazon.titan-text-lite-v1":{"mode":"chat","base_model":"titan-text-lite-v1","max_input_tokens":42000,"max_output_tokens":4000,"max_tokens":4000,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-east-1/amazon.titan-text-premier-v1:0":{"mode":"chat","base_model":"titan-text-premier","max_input_tokens":42000,"max_output_tokens":32000,"max_tokens":32000,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-east-1/anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","deprecation_date":"2026-07-30","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-east-1/anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","deprecation_date":"2026-09-10","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","ca-central-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-east-1/claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-east-1/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8000,"max_output_tokens":2048,"max_tokens":2048,"supports_pdf_input":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":2048}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/us-gov-east-1/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8000,"max_output_tokens":2048,"max_tokens":2048,"supports_pdf_input":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/amazon.nova-pro-v1:0":{"mode":"chat","base_model":"nova-pro","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/amazon.titan-embed-text-v1":{"mode":"embedding","base_model":"titan-embed-text-v1","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":1536,"provider":"bedrock"},"bedrock/us-gov-west-1/amazon.titan-embed-text-v2:0":{"mode":"embedding","base_model":"titan-embed-text","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":1024,"provider":"bedrock"},"bedrock/us-gov-west-1/amazon.titan-text-express-v1":{"mode":"chat","base_model":"titan-text-express-v1","max_input_tokens":42000,"max_output_tokens":8000,"max_tokens":8000,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/amazon.titan-text-lite-v1":{"mode":"chat","base_model":"titan-text-lite-v1","max_input_tokens":42000,"max_output_tokens":4000,"max_tokens":4000,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/amazon.titan-text-premier-v1:0":{"mode":"chat","base_model":"titan-text-premier","max_input_tokens":42000,"max_output_tokens":32000,"max_tokens":32000,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/anthropic.claude-3-7-sonnet-20250219-v1:0":{"mode":"chat","base_model":"claude-3-7-sonnet","provider":"bedrock","supports_prompt_caching":true,"source":"https://aws.amazon.com/bedrock/pricing/","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","deprecation_date":"2026-07-30","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","deprecation_date":"2026-09-10","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","ca-central-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-gov-west-1/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8000,"max_output_tokens":2048,"max_tokens":2048,"supports_pdf_input":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":2048}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/us-gov-west-1/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8000,"max_output_tokens":2048,"max_tokens":2048,"supports_pdf_input":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-1/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/us-west-1/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/1-month-commitment/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/1-month-commitment/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/1-month-commitment/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/6-month-commitment/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/6-month-commitment/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/6-month-commitment/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/anthropic.claude-instant-v1":{"mode":"chat","base_model":"claude-instant-v1","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/anthropic.claude-v1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/anthropic.claude-v2:1":{"mode":"chat","base_model":"claude","max_input_tokens":100000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/mistral.mistral-7b-instruct-v0:2":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8191}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"bedrock/us-west-2/mistral.mistral-large-2402-v1:0":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/mistral.mixtral-8x7b-instruct-v0:1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/moonshotai.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_response_schema":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us-west-2/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock/us.anthropic.claude-3-5-haiku-20241022-v1:0":{"mode":"chat","base_model":"claude-3-5-haiku","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"prompt_cache_min_tokens":2048,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2"],"deprecation_date":"2026-06-19","is_deprecated":true,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cerebras/llama-3.3-70b":{"mode":"chat","base_model":"llama-3.3-70b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"cerebras","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"cerebras/llama3.1-70b":{"mode":"chat","base_model":"llama-3.1-70b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"cerebras","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"cerebras/llama3.1-8b":{"mode":"chat","base_model":"llama-3.1-8b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"cerebras","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"cerebras/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.cerebras.ai/blog/openai-gpt-oss-120b-runs-fastest-on-cerebras","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"cerebras","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32768}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"cerebras/qwen-3-32b":{"mode":"chat","base_model":"qwen3-32b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://inference-docs.cerebras.ai/support/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"cerebras","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"cerebras/zai-glm-4.6":{"mode":"chat","base_model":"glm-4.6","deprecation_date":"2026-01-20","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://www.cerebras.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"cerebras","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"cerebras/zai-glm-4.7":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://www.cerebras.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"cerebras","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chat-bison":{"mode":"chat","base_model":"chat-bison","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chat-bison-32k":{"mode":"chat","base_model":"chat-bison-32k","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chat-bison-32k@002":{"mode":"chat","base_model":"chat-bison-32k","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chat-bison@001":{"mode":"chat","base_model":"chat-bison","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chat-bison@002":{"mode":"chat","base_model":"chat-bison","deprecation_date":"2025-04-09","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"chatdolphin":{"mode":"chat","base_model":"chatdolphin","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"provider":"nlp_cloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chatgpt-4o-latest":{"mode":"chat","base_model":"chatgpt-4o","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":4096}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"gpt-4o-transcribe-diarize":{"mode":"audio_transcription","base_model":"gpt-4o-transcribe","max_input_tokens":16000,"max_output_tokens":2000,"supported_endpoints":["/v1/audio/transcriptions"],"deprecation_date":"2027-02-26","source":"https://developers.openai.com/api/docs/pricing","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"claude-haiku-4-5-20251001":{"mode":"chat","base_model":"claude-haiku-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_structured_output":true,"supports_computer_use":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"provider":"anthropic","deprecation_date":"2026-10-15","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["auto","standard_only"],"supports_adaptive_thinking":false,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false,"supports_web_search":true},"claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_structured_output":true,"supports_computer_use":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"source":"https://platform.claude.com/docs/en/about-claude/pricing","provider":"anthropic","deprecation_date":"2026-10-15","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["auto","standard_only"],"supports_adaptive_thinking":false,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"claude-3-haiku-20240307":{"mode":"chat","base_model":"claude-3-haiku","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":264,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"deprecation_date":"2026-04-20","is_deprecated":true},"claude-4-opus-20250514":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"claude-4-sonnet-20250514":{"mode":"chat","base_model":"claude-sonnet-4","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"claude-sonnet-4-5":{"mode":"chat","base_model":"claude-sonnet-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"source":"https://platform.claude.com/docs/en/about-claude/pricing","provider":"anthropic","deprecation_date":"2026-09-29","tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["auto","standard_only"],"supports_adaptive_thinking":false,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false,"supports_web_search":true},"claude-sonnet-4-5-20250929":{"mode":"chat","base_model":"claude-sonnet-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"prompt_cache_min_tokens":1024,"source":"https://docs.anthropic.com/en/docs/about-claude/pricing","provider":"anthropic","deprecation_date":"2026-09-29","tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["auto","standard_only"],"supports_adaptive_thinking":false,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"claude-opus-4-1":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"anthropic","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":32000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"deprecation_date":"2026-08-05","is_deprecated":true,"prompt_cache_min_tokens":1024,"supports_adaptive_thinking":false,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_native_structured_output":true,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_web_search":true},"claude-opus-4-1-20250805":{"mode":"chat","base_model":"claude-opus-4-1","deprecation_date":"2026-08-05","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"is_deprecated":true},"claude-opus-4-5-20251101":{"mode":"chat","base_model":"claude-opus-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_thinking_cache_preservation":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_vision":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"provider":"anthropic","deprecation_date":"2026-11-24","tool_use_system_prompt_tokens":159,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high"],"service_tiers":["auto","standard_only"],"supports_adaptive_thinking":false,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false,"supports_web_search":true},"claude-opus-4-5":{"mode":"chat","base_model":"claude-opus-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_thinking_cache_preservation":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_vision":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"source":"https://platform.claude.com/docs/en/about-claude/pricing","provider":"anthropic","deprecation_date":"2026-11-24","supports_web_search":true,"tool_use_system_prompt_tokens":159,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high"],"service_tiers":["auto","standard_only"],"supports_adaptive_thinking":false,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"claude-opus-4-6":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider_specific_entry":{"us":1.1,"fast":6},"provider":"anthropic","supports_web_search":true,"deprecation_date":"2027-02-05","inference_geo_us_multiplier":1.1,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["auto","standard_only"],"supports_adaptive_thinking":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":true},"fast/claude-opus-4-6":{"mode":"chat","base_model":"fast/claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_web_search":true},"us/claude-opus-4-6":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_web_search":true},"fast/us/claude-opus-4-6":{"mode":"chat","base_model":"fast/us/claude-opus-4-6","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_web_search":true},"fast/claude-opus-4-6-20260205":{"mode":"chat","base_model":"fast/claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_web_search":true},"us/claude-opus-4-6-20260205":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_web_search":true},"claude-sonnet-4-20250514":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-06-15","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"is_deprecated":true},"cloudflare/@cf/meta/llama-2-7b-chat-fp16":{"mode":"chat","base_model":"","max_input_tokens":3072,"max_output_tokens":3072,"max_tokens":3072,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":3072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cloudflare/@cf/meta/llama-2-7b-chat-int8":{"mode":"chat","base_model":"","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cloudflare/@cf/mistral/mistral-7b-instruct-v0.1":{"mode":"chat","base_model":"","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cloudflare/@hf/thebloke/codellama-7b-instruct-awq":{"mode":"chat","base_model":"","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-bison":{"mode":"chat","base_model":"code-bison","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-bison-32k@002":{"mode":"completion","base_model":"code-bison-32k","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-bison32k":{"mode":"completion","base_model":"code-bison32k","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-bison@001":{"mode":"completion","base_model":"code-bison","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-bison@002":{"mode":"completion","base_model":"code-bison","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-gecko":{"mode":"completion","base_model":"code-gecko","max_input_tokens":2048,"max_output_tokens":64,"max_tokens":64,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-gecko-latest":{"mode":"completion","base_model":"code-gecko","max_input_tokens":2048,"max_output_tokens":64,"max_tokens":64,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-gecko@001":{"mode":"completion","base_model":"code-gecko","max_input_tokens":2048,"max_output_tokens":64,"max_tokens":64,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"code-gecko@002":{"mode":"completion","base_model":"code-gecko","max_input_tokens":2048,"max_output_tokens":64,"max_tokens":64,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codechat-bison":{"mode":"chat","base_model":"codechat-bison","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codechat-bison-32k":{"mode":"chat","base_model":"codechat-bison-32k","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codechat-bison-32k@002":{"mode":"chat","base_model":"codechat-bison-32k","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codechat-bison@001":{"mode":"chat","base_model":"codechat-bison","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codechat-bison@002":{"mode":"chat","base_model":"codechat-bison","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codechat-bison@latest":{"mode":"chat","base_model":"codechat-bison","max_input_tokens":6144,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codestral/codestral-2405":{"mode":"chat","base_model":"codestral","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://docs.mistral.ai/capabilities/code_generation/","supports_assistant_prefill":true,"supports_tool_choice":true,"provider":"codestral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codestral/codestral-latest":{"mode":"chat","base_model":"codestral","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://docs.mistral.ai/capabilities/code_generation/","supports_assistant_prefill":true,"supports_tool_choice":true,"provider":"codestral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"codex-mini-latest":{"mode":"responses","base_model":"codex-mini","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cohere.command-light-text-v14":{"mode":"chat","base_model":"command-light-text-v14","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cohere.command-r-plus-v1:0":{"mode":"chat","base_model":"command-r-plus","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cohere.command-r-v1:0":{"mode":"chat","base_model":"command-r","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cohere.command-text-v14":{"mode":"chat","base_model":"command-text-v14","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cohere.embed-english-v3":{"mode":"embedding","base_model":"embed-english-v3","max_input_tokens":512,"max_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","supports_embedding_image_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-west-2","ca-central-1","eu-central-1","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"]},"cohere.embed-multilingual-v3":{"mode":"embedding","base_model":"embed-multilingual-v3","max_input_tokens":512,"max_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","supports_embedding_image_input":true,"provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-west-2","ca-central-1","eu-central-1","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"]},"cohere.embed-v4:0":{"mode":"embedding","base_model":"embed","max_input_tokens":128000,"max_tokens":128000,"output_vector_size":1536,"source":"https://aws.amazon.com/bedrock/pricing/","supports_embedding_image_input":true,"provider":"bedrock","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","sa-east-1"]},"cohere/embed-v4.0":{"mode":"embedding","base_model":"embed","max_input_tokens":128000,"max_tokens":128000,"output_vector_size":1536,"supports_embedding_image_input":true,"provider":"cohere"},"cohere.rerank-v3-5:0":{"mode":"rerank","base_model":"rerank-v3-5","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"provider":"bedrock","max_document_chunks_per_query":100,"max_query_tokens":32000,"max_tokens_per_document_chunk":512,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"command":{"mode":"completion","base_model":"command","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"cohere","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4096}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":4000,"step":1}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"command-a-03-2025":{"mode":"chat","base_model":"command-a","max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"command-light":{"mode":"chat","base_model":"command-light","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"cohere","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4096}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":4000,"step":1}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"command-nightly":{"mode":"completion","base_model":"command-nightly","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":4096}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":128000,"step":10}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"command-r":{"mode":"chat","base_model":"command-r","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4096}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":128000,"step":10}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"command-r-08-2024":{"mode":"chat","base_model":"command-r","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4096}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":128000,"step":10}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"command-r-plus":{"mode":"chat","base_model":"command-r-plus","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4096}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":128000,"step":10}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"command-r-plus-08-2024":{"mode":"chat","base_model":"command-r-plus","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4096}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":128000,"step":10}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"command-r7b-12-2024":{"mode":"chat","base_model":"command-r7b","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://docs.cohere.com/v2/docs/command-r7b","supports_function_calling":true,"supports_tool_choice":true,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"computer-use-preview":{"mode":"chat","base_model":"computer-use","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://platform.openai.com/docs/models/computer-use-preview","provider":"azure","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dall-e-2":{"mode":"image_generation","base_model":"dall-e-2","supported_endpoints":["/v1/images/generations","/v1/images/edits","/v1/images/variations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek-chat":{"mode":"chat","base_model":"deepseek-chat","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"deepseek","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek-reasoner":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_function_calling":false,"supports_native_streaming":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"provider":"deepseek","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-coder":{"mode":"chat","base_model":"qwen-coder","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-flash":{"mode":"chat","base_model":"qwen-flash","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"range":[0,256000]},{"input_cost_per_token":2.5e-7,"output_cost_per_token":0.000002,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-flash-2025-07-28":{"mode":"chat","base_model":"qwen-flash","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"range":[0,256000]},{"input_cost_per_token":2.5e-7,"output_cost_per_token":0.000002,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-max":{"mode":"chat","base_model":"qwen-max","max_input_tokens":30720,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-plus":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-plus-2025-01-25":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-plus-2025-04-28":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-plus-2025-07-14":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-plus-2025-07-28":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-plus-2025-09-11":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-plus-latest":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-turbo":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-turbo-2024-11-01":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-turbo-2025-04-28":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen-turbo-latest":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen3-30b-a3b":{"mode":"chat","base_model":"qwen3-30b-a3b","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"dashscope/qwen3-coder-flash":{"mode":"chat","base_model":"qwen3-coder-flash","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"cache_read_input_token_cost":8e-8,"input_cost_per_token":3e-7,"output_cost_per_token":0.0000015,"range":[0,32000]},{"cache_read_input_token_cost":1.2e-7,"input_cost_per_token":5e-7,"output_cost_per_token":0.0000025,"range":[32000,128000]},{"cache_read_input_token_cost":2e-7,"input_cost_per_token":8e-7,"output_cost_per_token":0.000004,"range":[128000,256000]},{"cache_read_input_token_cost":4e-7,"input_cost_per_token":0.0000016,"output_cost_per_token":0.0000096,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen3-coder-flash-2025-07-28":{"mode":"chat","base_model":"qwen3-coder-flash","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":3e-7,"output_cost_per_token":0.0000015,"range":[0,32000]},{"input_cost_per_token":5e-7,"output_cost_per_token":0.0000025,"range":[32000,128000]},{"input_cost_per_token":8e-7,"output_cost_per_token":0.000004,"range":[128000,256000]},{"input_cost_per_token":0.0000016,"output_cost_per_token":0.0000096,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen3-coder-plus":{"mode":"chat","base_model":"qwen3-coder-plus","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"cache_read_input_token_cost":1e-7,"input_cost_per_token":0.000001,"output_cost_per_token":0.000005,"range":[0,32000]},{"cache_read_input_token_cost":1.8e-7,"input_cost_per_token":0.0000018,"output_cost_per_token":0.000009,"range":[32000,128000]},{"cache_read_input_token_cost":3e-7,"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,256000]},{"cache_read_input_token_cost":6e-7,"input_cost_per_token":0.000006,"output_cost_per_token":0.00006,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen3-coder-plus-2025-07-22":{"mode":"chat","base_model":"qwen3-coder-plus","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.000001,"output_cost_per_token":0.000005,"range":[0,32000]},{"input_cost_per_token":0.0000018,"output_cost_per_token":0.000009,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,256000]},{"input_cost_per_token":0.000006,"output_cost_per_token":0.00006,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen3-max-preview":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwen3-max":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dashscope/qwq-plus":{"mode":"chat","base_model":"qwq-plus","max_input_tokens":98304,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-bge-large-en":{"mode":"embedding","base_model":"bge-large-en","max_input_tokens":512,"max_tokens":512,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"output_vector_size":1024,"source":"https://www.databricks.com/product/pricing/foundation-model-serving","provider":"databricks"},"databricks/databricks-claude-3-7-sonnet":{"mode":"chat","base_model":"claude-3-7-sonnet","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"prompt_cache_min_tokens":4096,"provider":"databricks","supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-claude-opus-4":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"prompt_cache_min_tokens":1024,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-claude-opus-4-1":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"prompt_cache_min_tokens":1024,"provider":"databricks","supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-claude-opus-4-5":{"mode":"chat","base_model":"claude-opus-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"provider":"databricks","supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-claude-sonnet-4":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-10-09","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"prompt_cache_min_tokens":1024,"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"provider":"databricks","supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-claude-sonnet-4-1":{"mode":"chat","base_model":"claude-sonnet-4-1","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-claude-sonnet-4-5":{"mode":"chat","base_model":"claude-sonnet-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"prompt_cache_min_tokens":1024,"provider":"databricks","supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-gemini-2-5-flash":{"mode":"chat","base_model":"gemini-2.5-flash","deprecation_date":"2026-10-02","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_anthropic_thinking_payload":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_reasoning":true,"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_vision":true,"supports_audio_input":true,"supports_video_input":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-gemini-2-5-pro":{"mode":"chat","base_model":"gemini-2.5-pro","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_anthropic_thinking_payload":true,"supports_prompt_caching":true,"supports_tool_choice":true,"deprecation_date":"2026-10-02","supports_reasoning":true,"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_vision":true,"supports_audio_input":true,"supports_video_input":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-gemma-3-12b":{"mode":"chat","base_model":"gemma-3-12b","max_input_tokens":128000,"max_output_tokens":32000,"max_tokens":32000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","provider":"databricks","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"databricks/databricks-gpt-5":{"mode":"chat","base_model":"gpt-5","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"databricks","supports_assistant_prefill":true,"supports_reasoning":true,"supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-gpt-5-1":{"mode":"chat","base_model":"gpt-5-1","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"databricks","supports_assistant_prefill":true,"supports_reasoning":true,"supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"databricks","supports_assistant_prefill":true,"supports_reasoning":true,"supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"databricks","supports_assistant_prefill":true,"supports_reasoning":true,"supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","provider":"databricks","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"databricks/databricks-gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","provider":"databricks","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"databricks/databricks-gte-large-en":{"mode":"embedding","base_model":"gte-large-en","max_input_tokens":8192,"max_tokens":8192,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"output_vector_size":1024,"source":"https://www.databricks.com/product/pricing/foundation-model-serving","provider":"databricks"},"databricks/databricks-llama-2-70b-chat":{"mode":"chat","base_model":"llama-2-70b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"databricks/databricks-llama-4-maverick":{"mode":"chat","base_model":"llama-4-maverick","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Databricks documentation now provides both DBU costs (_dbu_cost_per_token) and dollar costs(_cost_per_token)."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-meta-llama-3-1-405b-instruct":{"mode":"chat","base_model":"llama-3-1-405b-instruct","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-meta-llama-3-1-8b-instruct":{"mode":"chat","base_model":"llama-3-1-8b-instruct","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-meta-llama-3-3-70b-instruct":{"mode":"chat","base_model":"llama-3-3-70b-instruct","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supports_tool_choice":true,"provider":"databricks","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_vision":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-meta-llama-3-70b-instruct":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"databricks/databricks-mixtral-8x7b-instruct":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-mpt-30b-instruct":{"mode":"chat","base_model":"mpt-30b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"databricks/databricks-mpt-7b-instruct":{"mode":"chat","base_model":"mpt-7b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070, based on databricks Llama 3.1 70B conversion. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dataforseo/search":{"mode":"search","base_model":"search","provider":"dataforseo","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"davinci-002":{"mode":"completion","base_model":"davinci-002","deprecation_date":"2026-09-28","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","provider":"text-completion-openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":false,"supports_tool_choice":false,"supports_parallel_function_calling":false,"supports_reasoning":false,"supports_response_schema":false,"supports_prompt_caching":false,"supports_web_search":false,"supports_service_tier":false,"supports_assistant_prefill":false,"supports_system_messages":false},"deepgram/base":{"mode":"audio_transcription","base_model":"base","metadata":{"calculation":"$0.0125/60 seconds = $0.00020833 per second","original_pricing_per_minute":0.0125},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/base-conversationalai":{"mode":"audio_transcription","base_model":"base-conversationalai","metadata":{"calculation":"$0.0125/60 seconds = $0.00020833 per second","original_pricing_per_minute":0.0125},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/base-finance":{"mode":"audio_transcription","base_model":"base-finance","metadata":{"calculation":"$0.0125/60 seconds = $0.00020833 per second","original_pricing_per_minute":0.0125},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/base-general":{"mode":"audio_transcription","base_model":"base-general","metadata":{"calculation":"$0.0125/60 seconds = $0.00020833 per second","original_pricing_per_minute":0.0125},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/base-meeting":{"mode":"audio_transcription","base_model":"base-meeting","metadata":{"calculation":"$0.0125/60 seconds = $0.00020833 per second","original_pricing_per_minute":0.0125},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/base-phonecall":{"mode":"audio_transcription","base_model":"base-phonecall","metadata":{"calculation":"$0.0125/60 seconds = $0.00020833 per second","original_pricing_per_minute":0.0125},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/base-video":{"mode":"audio_transcription","base_model":"base-video","metadata":{"calculation":"$0.0125/60 seconds = $0.00020833 per second","original_pricing_per_minute":0.0125},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/base-voicemail":{"mode":"audio_transcription","base_model":"base-voicemail","metadata":{"calculation":"$0.0125/60 seconds = $0.00020833 per second","original_pricing_per_minute":0.0125},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/enhanced":{"mode":"audio_transcription","base_model":"enhanced","metadata":{"calculation":"$0.0145/60 seconds = $0.00024167 per second","original_pricing_per_minute":0.0145},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/enhanced-finance":{"mode":"audio_transcription","base_model":"enhanced-finance","metadata":{"calculation":"$0.0145/60 seconds = $0.00024167 per second","original_pricing_per_minute":0.0145},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/enhanced-general":{"mode":"audio_transcription","base_model":"enhanced-general","metadata":{"calculation":"$0.0145/60 seconds = $0.00024167 per second","original_pricing_per_minute":0.0145},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/enhanced-meeting":{"mode":"audio_transcription","base_model":"enhanced-meeting","metadata":{"calculation":"$0.0145/60 seconds = $0.00024167 per second","original_pricing_per_minute":0.0145},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/enhanced-phonecall":{"mode":"audio_transcription","base_model":"enhanced-phonecall","metadata":{"calculation":"$0.0145/60 seconds = $0.00024167 per second","original_pricing_per_minute":0.0145},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova":{"mode":"audio_transcription","base_model":"nova","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2":{"mode":"audio_transcription","base_model":"nova-2","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-atc":{"mode":"audio_transcription","base_model":"nova-2-atc","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-automotive":{"mode":"audio_transcription","base_model":"nova-2-automotive","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-conversationalai":{"mode":"audio_transcription","base_model":"nova-2-conversationalai","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-drivethru":{"mode":"audio_transcription","base_model":"nova-2-drivethru","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-finance":{"mode":"audio_transcription","base_model":"nova-2-finance","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-general":{"mode":"audio_transcription","base_model":"nova-2-general","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-meeting":{"mode":"audio_transcription","base_model":"nova-2-meeting","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-phonecall":{"mode":"audio_transcription","base_model":"nova-2-phonecall","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-video":{"mode":"audio_transcription","base_model":"nova-2-video","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-2-voicemail":{"mode":"audio_transcription","base_model":"nova-2-voicemail","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-3":{"mode":"audio_transcription","base_model":"nova-3","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-3-general":{"mode":"audio_transcription","base_model":"nova-3-general","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-3-medical":{"mode":"audio_transcription","base_model":"nova-3-medical","metadata":{"calculation":"$0.0052/60 seconds = $0.00008667 per second (multilingual)","original_pricing_per_minute":0.0052},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-general":{"mode":"audio_transcription","base_model":"nova-general","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/nova-phonecall":{"mode":"audio_transcription","base_model":"nova-phonecall","metadata":{"calculation":"$0.0043/60 seconds = $0.00007167 per second","original_pricing_per_minute":0.0043},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/whisper":{"mode":"audio_transcription","base_model":"whisper","metadata":{"notes":"Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/whisper-base":{"mode":"audio_transcription","base_model":"whisper-base","metadata":{"notes":"Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/whisper-large":{"mode":"audio_transcription","base_model":"whisper-large","metadata":{"notes":"Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/whisper-medium":{"mode":"audio_transcription","base_model":"whisper-medium","metadata":{"notes":"Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/whisper-small":{"mode":"audio_transcription","base_model":"whisper-small","metadata":{"notes":"Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepgram/whisper-tiny":{"mode":"audio_transcription","base_model":"whisper-tiny","metadata":{"notes":"Deepgram's hosted OpenAI Whisper models - pricing may differ from native Deepgram models"},"source":"https://deepgram.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"deepgram","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Gryphe/MythoMax-L2-13b":{"mode":"chat","base_model":"mythomax-l2-13b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/NousResearch/Hermes-3-Llama-3.1-405B":{"mode":"chat","base_model":"hermes-3-llama-3.1-405b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/NousResearch/Hermes-3-Llama-3.1-70B":{"mode":"chat","base_model":"hermes-3-llama-3.1-70b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Qwen/QwQ-32B":{"mode":"chat","base_model":"qwq-32b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/Qwen/Qwen2.5-72B-Instruct":{"mode":"chat","base_model":"qwen2.5-72b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/Qwen/Qwen2.5-7B-Instruct":{"mode":"chat","base_model":"qwen2.5-7b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/Qwen/Qwen2.5-VL-32B-Instruct":{"mode":"chat","base_model":"qwen2.5-vl-32b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_tool_choice":true,"supports_vision":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Qwen/Qwen3-14B":{"mode":"chat","base_model":"qwen3-14b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":40960}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/Qwen/Qwen3-235B-A22B":{"mode":"chat","base_model":"qwen3-235b-a22b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":40960}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Qwen/Qwen3-30B-A3B":{"mode":"chat","base_model":"qwen3-30b-a3b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":40960}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/Qwen/Qwen3-32B":{"mode":"chat","base_model":"qwen3-32b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":40960}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct-turbo","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_tool_choice":true,"supports_function_calling":true,"supports_prompt_caching":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Qwen/Qwen3-Next-80B-A3B-Instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Qwen/Qwen3-Next-80B-A3B-Thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Sao10K/L3-8B-Lunaris-v1-Turbo":{"mode":"chat","base_model":"l3-8b-lunaris-v1-turbo","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Sao10K/L3.1-70B-Euryale-v2.2":{"mode":"chat","base_model":"l3.1-70b-euryale","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/Sao10K/L3.3-70B-Euryale-v2.3":{"mode":"chat","base_model":"l3.3-70b-euryale","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/allenai/olmOCR-7B-0725-FP8":{"mode":"chat","base_model":"olmocr-7b-0725-fp8","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/anthropic/claude-3-7-sonnet-latest":{"mode":"chat","base_model":"claude-3-7-sonnet","max_tokens":200000,"max_input_tokens":200000,"max_output_tokens":200000,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/anthropic/claude-4-opus":{"mode":"chat","base_model":"claude-opus-4","max_tokens":200000,"max_input_tokens":200000,"max_output_tokens":200000,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/anthropic/claude-4-sonnet":{"mode":"chat","base_model":"claude-sonnet-4","max_tokens":200000,"max_input_tokens":200000,"max_output_tokens":200000,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/deepseek-ai/DeepSeek-R1":{"mode":"chat","base_model":"deepseek-r1","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/deepseek-ai/DeepSeek-R1-0528":{"mode":"chat","base_model":"deepseek-r1","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/deepseek-ai/DeepSeek-R1-0528-Turbo":{"mode":"chat","base_model":"deepseek-r1-0528-turbo","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Llama-70B":{"mode":"chat","base_model":"deepseek-r1-distill-llama-70b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-32b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/deepseek-ai/DeepSeek-R1-Turbo":{"mode":"chat","base_model":"deepseek-r1-turbo","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/deepseek-ai/DeepSeek-V3":{"mode":"chat","base_model":"deepseek-v3","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/deepseek-ai/DeepSeek-V3-0324":{"mode":"chat","base_model":"deepseek-v3","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"supports_tool_choice":true,"supports_function_calling":true,"supports_prompt_caching":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/deepseek-ai/DeepSeek-V3.1":{"mode":"chat","base_model":"deepseek-v3.1","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"supports_tool_choice":true,"supports_reasoning":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":163840,"range":{"min":1,"max":163840}}]},"deepinfra/deepseek-ai/DeepSeek-V3.1-Terminus":{"mode":"chat","base_model":"deepseek-v3.1-terminus","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/google/gemini-2.0-flash-001":{"mode":"chat","base_model":"gemini-2.0-flash","deprecation_date":"2026-03-31","max_tokens":1000000,"max_input_tokens":1000000,"max_output_tokens":1000000,"supports_tool_choice":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"deepinfra/google/gemini-2.5-flash":{"mode":"chat","base_model":"gemini-2.5-flash","max_tokens":1000000,"max_input_tokens":1000000,"max_output_tokens":1000000,"supports_tool_choice":true,"supports_function_calling":true,"supports_image_size":false,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_audio_input":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/google/gemini-2.5-pro":{"mode":"chat","base_model":"gemini-2.5-pro","max_tokens":1000000,"max_input_tokens":1000000,"max_output_tokens":1000000,"supports_tool_choice":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_audio_input":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/google/gemma-3-12b-it":{"mode":"chat","base_model":"gemma-3-12b-it","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/google/gemma-3-27b-it":{"mode":"chat","base_model":"gemma-3-27b-it","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/google/gemma-3-4b-it":{"mode":"chat","base_model":"gemma-3-4b-it","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct":{"mode":"chat","base_model":"llama-3.2-11b-vision-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Llama-3.2-3B-Instruct":{"mode":"chat","base_model":"llama-3.2-3b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Llama-3.3-70B-Instruct-Turbo":{"mode":"chat","base_model":"llama-3.3-70b-instruct-turbo","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct-fp8","max_tokens":1048576,"max_input_tokens":1048576,"max_output_tokens":1048576,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_tokens":327680,"max_input_tokens":327680,"max_output_tokens":327680,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":327680}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/meta-llama/Llama-Guard-3-8B":{"mode":"chat","base_model":"llama-guard-3-8b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/meta-llama/Llama-Guard-4-12B":{"mode":"chat","base_model":"llama-guard-4-12b","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Meta-Llama-3-8B-Instruct":{"mode":"chat","base_model":"llama-3-8b-instruct","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct":{"mode":"chat","base_model":"llama-3.1-70b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo":{"mode":"chat","base_model":"llama-3.1-70b-instruct-turbo","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo":{"mode":"chat","base_model":"llama-3.1-8b-instruct-turbo","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/microsoft/WizardLM-2-8x22B":{"mode":"chat","base_model":"wizardlm-2-8x22b","max_tokens":65536,"max_input_tokens":65536,"max_output_tokens":65536,"supports_tool_choice":false,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":65536}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/microsoft/phi-4":{"mode":"chat","base_model":"phi-4","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/mistralai/Mistral-Nemo-Instruct-2407":{"mode":"chat","base_model":"mistral-nemo-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/mistralai/Mistral-Small-24B-Instruct-2501":{"mode":"chat","base_model":"mistral-small-24b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/mistralai/Mistral-Small-3.2-24B-Instruct-2506":{"mode":"chat","base_model":"mistral-small-3.2-24b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/mistralai/Mixtral-8x7B-Instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/moonshotai/Kimi-K2-Instruct":{"mode":"chat","base_model":"kimi-k2-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/moonshotai/Kimi-K2-Instruct-0905":{"mode":"chat","base_model":"kimi-k2-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct":{"mode":"chat","base_model":"llama-3.1-nemotron-70b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/nvidia/Llama-3.3-Nemotron-Super-49B-v1.5":{"mode":"chat","base_model":"llama-3.3-nemotron-super-49b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/nvidia/NVIDIA-Nemotron-Nano-9B-v2":{"mode":"chat","base_model":"nvidia-nemotron-nano-9b-v2","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepinfra/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"deepinfra/zai-org/GLM-4.5":{"mode":"chat","base_model":"glm-4.5","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_tool_choice":true,"supports_function_calling":true,"provider":"deepinfra","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek/deepseek-chat":{"mode":"chat","base_model":"deepseek-chat","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"deepseek","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek/deepseek-coder":{"mode":"chat","base_model":"deepseek-coder","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"deepseek","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek/deepseek-r1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"deepseek","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek/deepseek-reasoner":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":false,"supports_native_streaming":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"provider":"deepseek","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek/deepseek-v3":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"deepseek","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek/deepseek-v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"deepseek","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek.v3-v1:0":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":163840,"max_output_tokens":81920,"max_tokens":81920,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_native_structured_output":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":81920}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"dolphin":{"mode":"completion","base_model":"dolphin","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"provider":"nlp_cloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"deepseek-v3-2-251201":{"mode":"chat","base_model":"deepseek-v3-2","max_input_tokens":98304,"max_output_tokens":32768,"max_tokens":32768,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"volcengine","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"glm-4-7-251222":{"mode":"chat","base_model":"glm-4-7","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"volcengine","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"kimi-k2-thinking-251104":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":229376,"max_output_tokens":32768,"max_tokens":32768,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"volcengine","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"doubao-embedding":{"mode":"embedding","base_model":"doubao-embedding","max_input_tokens":4096,"max_tokens":4096,"metadata":{"notes":"Volcengine Doubao embedding model - standard version with 2560 dimensions"},"output_vector_size":2560,"provider":"volcengine"},"doubao-embedding-large":{"mode":"embedding","base_model":"doubao-embedding-large","max_input_tokens":4096,"max_tokens":4096,"metadata":{"notes":"Volcengine Doubao embedding model - large version with 2048 dimensions"},"output_vector_size":2048,"provider":"volcengine"},"doubao-embedding-large-text-240915":{"mode":"embedding","base_model":"doubao-embedding-large-text","max_input_tokens":4096,"max_tokens":4096,"metadata":{"notes":"Volcengine Doubao embedding model - text-240915 version with 4096 dimensions"},"output_vector_size":4096,"provider":"volcengine"},"doubao-embedding-large-text-250515":{"mode":"embedding","base_model":"doubao-embedding-large-text","max_input_tokens":4096,"max_tokens":4096,"metadata":{"notes":"Volcengine Doubao embedding model - text-250515 version with 2048 dimensions"},"output_vector_size":2048,"provider":"volcengine"},"doubao-embedding-text-240715":{"mode":"embedding","base_model":"doubao-embedding-text","max_input_tokens":4096,"max_tokens":4096,"metadata":{"notes":"Volcengine Doubao embedding model - text-240715 version with 2560 dimensions"},"output_vector_size":2560,"provider":"volcengine"},"exa_ai/search":{"mode":"search","base_model":"search","tiered_pricing":[{"input_cost_per_query":0.005,"max_results_range":[0,25]},{"input_cost_per_query":0.025,"max_results_range":[26,100]}],"provider":"exa_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"firecrawl/search":{"mode":"search","base_model":"search","tiered_pricing":[{"input_cost_per_query":0.00166,"max_results_range":[1,10]},{"input_cost_per_query":0.00332,"max_results_range":[11,20]},{"input_cost_per_query":0.00498,"max_results_range":[21,30]},{"input_cost_per_query":0.00664,"max_results_range":[31,40]},{"input_cost_per_query":0.0083,"max_results_range":[41,50]},{"input_cost_per_query":0.00996,"max_results_range":[51,60]},{"input_cost_per_query":0.01162,"max_results_range":[61,70]},{"input_cost_per_query":0.01328,"max_results_range":[71,80]},{"input_cost_per_query":0.01494,"max_results_range":[81,90]},{"input_cost_per_query":0.0166,"max_results_range":[91,100]}],"metadata":{"notes":"Firecrawl search pricing: $83 for 100,000 credits, 2 credits per 10 results. Cost = ceiling(limit/10) * 2 * $0.00083"},"provider":"firecrawl","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/search":{"mode":"search","base_model":"search","provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"searxng/search":{"mode":"search","base_model":"search","metadata":{"notes":"SearXNG is an open-source metasearch engine. Free to use when self-hosted or using public instances."},"provider":"searxng","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"elevenlabs/scribe_v1":{"mode":"audio_transcription","base_model":"scribe-v1","metadata":{"calculation":"$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)","notes":"ElevenLabs Scribe v1 - state-of-the-art speech recognition model with 99 language support","original_pricing_per_hour":0.22},"source":"https://elevenlabs.io/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"elevenlabs","model_parameters":[{"id":"language","label":"Language","helpText":"The language of the audio to transcribe. If not specified, the model will auto-detect the language.","type":"select","default":"auto","options":[{"label":"Auto-detect","value":"auto"},{"label":"English","value":"en"},{"label":"Spanish","value":"es"},{"label":"French","value":"fr"},{"label":"German","value":"de"},{"label":"Italian","value":"it"},{"label":"Portuguese","value":"pt"},{"label":"Polish","value":"pl"},{"label":"Dutch","value":"nl"},{"label":"Japanese","value":"ja"},{"label":"Chinese","value":"zh"},{"label":"Korean","value":"ko"},{"label":"Russian","value":"ru"},{"label":"Arabic","value":"ar"},{"label":"Hindi","value":"hi"}]},{"id":"timestamp_granularities","label":"Timestamp Granularities","helpText":"The level of detail for timestamps in the transcription.","type":"select","default":"segment","options":[{"label":"Segment","value":"segment"},{"label":"Word","value":"word"}]},{"id":"diarization","label":"Speaker Diarization","helpText":"Enable speaker diarization to identify and separate different speakers in the audio.","type":"boolean","default":false},{"id":"num_speakers","label":"Number of Speakers","helpText":"Expected number of speakers in the audio (used when diarization is enabled).","type":"number","default":2,"range":{"min":1,"max":10,"step":1}},{"id":"response_format","label":"Response Format","helpText":"The format of the transcription output.","type":"select","default":"json","options":[{"label":"JSON","value":"json"},{"label":"Text","value":"text"},{"label":"SRT","value":"srt"},{"label":"VTT","value":"vtt"}]},{"id":"punctuation","label":"Punctuation","helpText":"Enable automatic punctuation in the transcription.","type":"boolean","default":true},{"id":"remove_disfluencies","label":"Remove Disfluencies","helpText":"Remove filler words like 'um', 'uh', stutters, and false starts from the transcription.","type":"boolean","default":false},{"id":"profanity_filter","label":"Profanity Filter","helpText":"Enable filtering of profane words in the transcription.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"elevenlabs/scribe_v1_experimental":{"mode":"audio_transcription","base_model":"scribe-v1","metadata":{"calculation":"$0.22/hour = $0.00366/minute = $0.0000611 per second (enterprise pricing)","notes":"ElevenLabs Scribe v1 experimental - enhanced version of the main Scribe model","original_pricing_per_hour":0.22},"source":"https://elevenlabs.io/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"elevenlabs","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"elevenlabs/eleven_v3":{"mode":"audio_speech","base_model":"eleven-v3","metadata":{"calculation":"$0.18/1000 characters (Scale plan pricing, 1 credit per character)","notes":"ElevenLabs Eleven v3 - most expressive TTS model with 70+ languages and audio tags support"},"source":"https://elevenlabs.io/pricing","supported_endpoints":["/v1/audio/speech"],"provider":"elevenlabs","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"elevenlabs/eleven_multilingual_v2":{"mode":"audio_speech","base_model":"eleven-multilingual-v2","metadata":{"calculation":"$0.18/1000 characters (Scale plan pricing, 1 credit per character)","notes":"ElevenLabs Eleven Multilingual v2 - default TTS model with 29 languages support"},"source":"https://elevenlabs.io/pricing","supported_endpoints":["/v1/audio/speech"],"provider":"elevenlabs","model_parameters":[{"id":"voice_id","label":"Voice","helpText":"The voice to use for speech synthesis. You can use pre-made voices or custom cloned voices.","type":"select","required":true},{"id":"stability","label":"Stability","helpText":"Controls the consistency and predictability of the voice. Lower values allow for more variable and expressive speech, while higher values make the voice more stable and consistent.","type":"number","default":0.5,"range":{"min":0,"max":1,"step":0.01}},{"id":"similarity_boost","label":"Similarity Boost","helpText":"Enhances the similarity to the original voice. Higher values make the generated speech closer to the original voice characteristics.","type":"number","default":0.75,"range":{"min":0,"max":1,"step":0.01}},{"id":"style","label":"Style Exaggeration","helpText":"Controls how much the AI should exaggerate the style of the voice. Higher values result in more expressive and dramatic speech.","type":"number","default":0,"range":{"min":0,"max":1,"step":0.01}},{"id":"use_speaker_boost","label":"Speaker Boost","helpText":"Boost the similarity to the speaker and reduces differences between speakers. Useful when using different voices in the same audio.","type":"boolean","default":true},{"id":"model_id","label":"Model","helpText":"The TTS model to use for generation.","type":"select","default":"eleven_multilingual_v2","options":[{"label":"Eleven Multilingual v2","value":"eleven_multilingual_v2"},{"label":"Eleven Turbo v2","value":"eleven_turbo_v2"},{"label":"Eleven Turbo v2.5","value":"eleven_turbo_v2_5"},{"label":"Eleven Monolingual v1","value":"eleven_monolingual_v1"}]},{"id":"language_code","label":"Language","helpText":"The language code for the text to be synthesized. Required for multilingual models to optimize pronunciation.","type":"select","default":"en","options":[{"label":"English","value":"en"},{"label":"Spanish","value":"es"},{"label":"French","value":"fr"},{"label":"German","value":"de"},{"label":"Italian","value":"it"},{"label":"Portuguese","value":"pt"},{"label":"Polish","value":"pl"},{"label":"Dutch","value":"nl"},{"label":"Japanese","value":"ja"},{"label":"Chinese","value":"zh"},{"label":"Korean","value":"ko"},{"label":"Russian","value":"ru"},{"label":"Arabic","value":"ar"},{"label":"Hindi","value":"hi"},{"label":"Turkish","value":"tr"},{"label":"Swedish","value":"sv"},{"label":"Indonesian","value":"id"},{"label":"Filipino","value":"fil"},{"label":"Ukrainian","value":"uk"},{"label":"Greek","value":"el"},{"label":"Czech","value":"cs"},{"label":"Finnish","value":"fi"},{"label":"Romanian","value":"ro"},{"label":"Danish","value":"da"},{"label":"Bulgarian","value":"bg"},{"label":"Malay","value":"ms"},{"label":"Slovak","value":"sk"},{"label":"Croatian","value":"hr"},{"label":"Tamil","value":"ta"},{"label":"Vietnamese","value":"vi"}]},{"id":"output_format","label":"Output Format","helpText":"The audio format for the generated speech.","type":"select","default":"mp3_44100_128","options":[{"label":"MP3 (44.1kHz, 128kbps)","value":"mp3_44100_128"},{"label":"MP3 (44.1kHz, 192kbps)","value":"mp3_44100_192"},{"label":"MP3 (22.05kHz, 32kbps)","value":"mp3_22050_32"},{"label":"PCM (16kHz, 16-bit)","value":"pcm_16000"},{"label":"PCM (22.05kHz, 16-bit)","value":"pcm_22050"},{"label":"PCM (24kHz, 16-bit)","value":"pcm_24000"},{"label":"PCM (44.1kHz, 16-bit)","value":"pcm_44100"},{"label":"��-law (8kHz, 8-bit)","value":"ulaw_8000"}]},{"id":"optimize_streaming_latency","label":"Optimize Streaming Latency","helpText":"Optimize the model for lower latency at the cost of quality. Higher values reduce latency but may decrease audio quality.","type":"number","default":0,"range":{"min":0,"max":4,"step":1}},{"id":"seed","label":"Seed","helpText":"Random seed for reproducible generation. If set, the same text with the same settings will produce identical audio.","type":"number"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"embed-english-light-v2.0":{"mode":"embedding","base_model":"embed-english-light","max_input_tokens":1024,"max_tokens":1024,"provider":"cohere","deprecation_date":"2026-04-04","is_deprecated":true},"embed-english-light-v3.0":{"mode":"embedding","base_model":"embed-english-light","max_input_tokens":1024,"max_tokens":1024,"provider":"cohere"},"embed-english-v2.0":{"mode":"embedding","base_model":"embed-english","max_input_tokens":4096,"max_tokens":4096,"provider":"cohere","deprecation_date":"2026-04-04","is_deprecated":true},"embed-english-v3.0":{"mode":"embedding","base_model":"embed-english","max_input_tokens":1024,"max_tokens":1024,"metadata":{"notes":"'supports_image_input' is a deprecated field. Use 'supports_embedding_image_input' instead."},"supports_embedding_image_input":true,"supports_image_input":true,"provider":"cohere"},"embed-multilingual-v2.0":{"mode":"embedding","base_model":"embed-multilingual","max_input_tokens":768,"max_tokens":768,"provider":"cohere","deprecation_date":"2026-04-04","is_deprecated":true},"embed-multilingual-v3.0":{"mode":"embedding","base_model":"embed-multilingual","max_input_tokens":1024,"max_tokens":1024,"supports_embedding_image_input":true,"provider":"cohere"},"embed-multilingual-light-v3.0":{"mode":"embedding","base_model":"embed-multilingual-light","max_input_tokens":1024,"max_tokens":1024,"supports_embedding_image_input":true,"provider":"cohere"},"eu.amazon.nova-lite-v1:0":{"mode":"chat","base_model":"nova-lite","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.amazon.nova-micro-v1:0":{"mode":"chat","base_model":"nova-micro","max_input_tokens":128000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.amazon.nova-pro-v1:0":{"mode":"chat","base_model":"nova-pro","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-3-5-haiku-20241022-v1:0":{"mode":"chat","base_model":"claude-3-5-haiku","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"prompt_cache_min_tokens":2048,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","deprecation_date":"2026-10-15","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"eu.anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-3-5-sonnet-20241022-v2:0":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-3-7-sonnet-20250219-v1:0":{"mode":"chat","base_model":"claude-3-7-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","deprecation_date":"2026-09-10","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","ca-central-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-3-opus-20240229-v1:0":{"mode":"chat","base_model":"claude-3-opus","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-3-sonnet-20240229-v1:0":{"mode":"chat","base_model":"claude-3-sonnet","deprecation_date":"2026-07-30","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-opus-4-1-20250805-v1:0":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"deprecation_date":"2027-01-08","provider":"bedrock","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-opus-4-20250514-v1:0":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-sonnet-4-20250514-v1:0":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-10-14","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"bedrock_converse_supports_strict_tools":false,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"eu.meta.llama3-2-1b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-1b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.meta.llama3-2-3b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-3b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.mistral.pixtral-large-2502-v1:0":{"mode":"chat","base_model":"pixtral-large","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/bria/text-to-image/3.2":{"mode":"image_generation","base_model":"bria/text-to-image/3.2","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/flux-pro/v1.1":{"mode":"image_generation","base_model":"flux-pro/v1.1","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/flux-pro/v1.1-ultra":{"mode":"image_generation","base_model":"flux-pro/v1.1-ultra","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/flux/schnell":{"mode":"image_generation","base_model":"flux/schnell","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/bytedance/seedream/v3/text-to-image":{"mode":"image_generation","base_model":"seedream/v3/text-to-image","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/bytedance/dreamina/v3.1/text-to-image":{"mode":"image_generation","base_model":"dreamina/v3.1/text-to-image","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/ideogram/v3":{"mode":"image_generation","base_model":"ideogram/v3","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/imagen4/preview":{"mode":"image_generation","base_model":"imagen4/preview","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/imagen4/preview/fast":{"mode":"image_generation","base_model":"imagen4/preview/fast","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/imagen4/preview/ultra":{"mode":"image_generation","base_model":"imagen4/preview/ultra","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/recraft/v3/text-to-image":{"mode":"image_generation","base_model":"v3/text-to-image","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fal_ai/fal-ai/stable-diffusion-v35-medium":{"mode":"image_generation","base_model":"stable-diffusion-v35-medium","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"featherless_ai/featherless-ai/Qwerky-72B":{"mode":"chat","base_model":"qwerky-72b","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"provider":"featherless_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"featherless_ai/featherless-ai/Qwerky-QwQ-32B":{"mode":"chat","base_model":"qwerky-qwq-32b","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"provider":"featherless_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks-ai-4.1b-to-16b":{"mode":"chat","base_model":"fireworks-ai-4.1b-to-16b","provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks-ai-56b-to-176b":{"mode":"chat","base_model":"fireworks-ai-56b-to-176b","provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks-ai-above-16b":{"mode":"chat","base_model":"fireworks-ai-above-16b","provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks-ai-default":{"mode":"chat","base_model":"fireworks-ai-default","provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks-ai-embedding-150m-to-350m":{"mode":"chat","base_model":"fireworks-ai-embedding-150m-to-350m","provider":"fireworks_ai-embedding-models","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks-ai-embedding-up-to-150m":{"mode":"chat","base_model":"fireworks-ai-embedding-up-to-150m","provider":"fireworks_ai-embedding-models","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks-ai-moe-up-to-56b":{"mode":"chat","base_model":"fireworks-ai-moe-up-to-56b","provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks-ai-up-to-4b":{"mode":"chat","base_model":"fireworks-ai-up-to-4b","provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/WhereIsAI/UAE-Large-V1":{"mode":"embedding","base_model":"uae-large-v1","max_input_tokens":512,"max_tokens":512,"source":"https://fireworks.ai/pricing","provider":"fireworks_ai-embedding-models"},"fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-instruct":{"mode":"chat","base_model":"deepseek-coder-v2-instruct","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"source":"https://fireworks.ai/pricing","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-r1":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/deepseek-r1","max_input_tokens":160000,"max_output_tokens":160000,"max_tokens":160000,"source":"merged_from_llm_models_csv","supports_response_schema":true,"supports_tool_choice":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":160000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_reasoning":true},"fireworks_ai/accounts/fireworks/models/deepseek-r1-0528":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":160000,"max_output_tokens":160000,"max_tokens":160000,"source":"https://fireworks.ai/pricing","supports_response_schema":true,"supports_tool_choice":false,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":160000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-r1-basic":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/deepseek-r1-basic","max_input_tokens":160000,"max_output_tokens":160000,"max_tokens":160000,"source":"merged_from_llm_models_csv","supports_response_schema":true,"supports_tool_choice":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":160000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_reasoning":true},"fireworks_ai/accounts/fireworks/models/deepseek-v3":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/deepseek-v3","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_response_schema":true,"supports_tool_choice":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true},"fireworks_ai/accounts/fireworks/models/deepseek-v3-0324":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/deepseek-v3-0324","max_input_tokens":160000,"max_output_tokens":160000,"max_tokens":160000,"source":"merged_from_llm_models_csv","supports_response_schema":true,"supports_tool_choice":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":160000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true},"fireworks_ai/accounts/fireworks/models/deepseek-v3p1":{"mode":"chat","base_model":"deepseek-v3.1","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://fireworks.ai/pricing","supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}}]},"fireworks_ai/accounts/fireworks/models/deepseek-v3p1-terminus":{"mode":"chat","base_model":"deepseek-v3.1-terminus","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://fireworks.ai/pricing","supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-v3p2":{"mode":"chat","base_model":"deepseek-v3.2","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"source":"https://fireworks.ai/models/fireworks/deepseek-v3p2","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/firefunction-v2":{"mode":"chat","base_model":"firefunction-v2","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://fireworks.ai/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/glm-4p5":{"mode":"chat","base_model":"glm-4.5","max_input_tokens":128000,"max_output_tokens":96000,"max_tokens":96000,"source":"https://fireworks.ai/models/fireworks/glm-4p5","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/glm-4p5-air":{"mode":"chat","base_model":"glm-4.5-air","max_input_tokens":128000,"max_output_tokens":96000,"max_tokens":96000,"source":"https://artificialanalysis.ai/models/glm-4-5-air","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/glm-4p6":{"mode":"chat","base_model":"glm-4.6","max_input_tokens":202800,"max_output_tokens":202800,"max_tokens":202800,"source":"https://fireworks.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":202800}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://fireworks.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://fireworks.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/kimi-k2-instruct":{"mode":"chat","base_model":"kimi-k2-instruct","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"source":"https://fireworks.ai/models/fireworks/kimi-k2-instruct","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/kimi-k2-instruct-0905":{"mode":"chat","base_model":"kimi-k2-instruct","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://app.fireworks.ai/models/fireworks/kimi-k2-instruct-0905","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://fireworks.ai/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/kimi-k2p5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://fireworks.ai/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v3p1-405b-instruct":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/llama-v3p1-405b-instruct","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v3p1-8b-instruct":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/llama-v3p1-8b-instruct","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v3p2-11b-vision-instruct":{"mode":"chat","base_model":"llama-3.2-11b-vision-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://fireworks.ai/pricing","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v3p2-1b-instruct":{"mode":"chat","base_model":"llama-3.2-1b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://fireworks.ai/pricing","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v3p2-3b-instruct":{"mode":"chat","base_model":"llama-3.2-3b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://fireworks.ai/pricing","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v3p2-90b-vision-instruct":{"mode":"chat","base_model":"llama-3.2-90b-vision-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://fireworks.ai/pricing","supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama4-maverick-instruct-basic":{"mode":"image_generation","base_model":"fireworks/accounts/fireworks/models/llama4-maverick-instruct-basic","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"merged_from_llm_models_csv","supports_response_schema":true,"supports_tool_choice":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"min_p","label":"Min P","helpText":"A number between 0 and 1 that can be used as an alternative to top_p and top-k","type":"number","default":0,"range":{"min":0,"max":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}]},"fireworks_ai/accounts/fireworks/models/llama4-scout-instruct-basic":{"mode":"image_generation","base_model":"fireworks/accounts/fireworks/models/llama4-scout-instruct-basic","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"merged_from_llm_models_csv","supports_response_schema":true,"supports_tool_choice":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"min_p","label":"Min P","helpText":"A number between 0 and 1 that can be used as an alternative to top_p and top-k","type":"number","default":0,"range":{"min":0,"max":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}]},"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct-hf":{"mode":"chat","base_model":"mixtral-8x22b-instruct","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"source":"https://fireworks.ai/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":65536}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2-72b-instruct":{"mode":"chat","base_model":"qwen2-72b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://fireworks.ai/pricing","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://fireworks.ai/pricing","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/yi-large":{"mode":"chat","base_model":"yi-large","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://fireworks.ai/pricing","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/nomic-ai/nomic-embed-text-v1":{"mode":"embedding","base_model":"nomic-embed-text-v1","max_input_tokens":8192,"max_tokens":8192,"source":"https://fireworks.ai/pricing","provider":"fireworks_ai-embedding-models"},"fireworks_ai/nomic-ai/nomic-embed-text-v1.5":{"mode":"embedding","base_model":"nomic-embed-text","max_input_tokens":8192,"max_tokens":8192,"source":"https://fireworks.ai/pricing","provider":"fireworks_ai-embedding-models"},"fireworks_ai/thenlper/gte-base":{"mode":"embedding","base_model":"gte-base","max_input_tokens":512,"max_tokens":512,"source":"https://fireworks.ai/pricing","provider":"fireworks_ai-embedding-models"},"fireworks_ai/thenlper/gte-large":{"mode":"embedding","base_model":"gte-large","max_input_tokens":512,"max_tokens":512,"source":"https://fireworks.ai/pricing","provider":"fireworks_ai-embedding-models"},"friendliai/meta-llama-3.1-70b-instruct":{"mode":"chat","base_model":"llama-3.1-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"friendliai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"friendliai/meta-llama-3.1-8b-instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"friendliai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ft:babbage-002":{"mode":"completion","base_model":"babbage-002","deprecation_date":"2026-10-23","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","provider":"text-completion-openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ft:davinci-002":{"mode":"completion","base_model":"davinci-002","deprecation_date":"2026-10-23","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","provider":"text-completion-openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ft:gpt-3.5-turbo":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2026-10-23","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ft:gpt-3.5-turbo-0125":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2026-10-23","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ft:gpt-3.5-turbo-0613":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2026-10-23","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ft:gpt-3.5-turbo-1106":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2026-10-23","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ft:gpt-4-0613":{"mode":"chat","base_model":"gpt-4","deprecation_date":"2026-10-23","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"OpenAI needs to add pricing for this ft model, will be updated when added by OpenAI. Defaulting to base model pricing","supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ft:gpt-4o-2024-08-06":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ft:gpt-4o-2024-11-20":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ft:gpt-4o-mini-2024-07-18":{"mode":"chat","base_model":"gpt-4o-mini","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ft:gpt-4.1-2025-04-14":{"mode":"chat","base_model":"gpt-4.1","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ft:gpt-4.1-mini-2025-04-14":{"mode":"chat","base_model":"gpt-4.1-mini","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ft:gpt-4.1-nano-2025-04-14":{"mode":"chat","base_model":"gpt-4.1-nano","deprecation_date":"2026-10-23","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ft:o4-mini-2025-04-16":{"mode":"chat","base_model":"o4-mini","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"gemini-1.0-pro":{"mode":"chat","base_model":"gemini-1.0-pro","max_input_tokens":32760,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#google_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-1.0-pro-001":{"mode":"chat","base_model":"gemini-1.0-pro","deprecation_date":"2025-04-09","max_input_tokens":32760,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.0-pro-002":{"mode":"chat","base_model":"gemini-1.0-pro","deprecation_date":"2025-04-09","max_input_tokens":32760,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.0-pro-vision":{"mode":"chat","base_model":"gemini-1.0-pro-vision","max_images_per_prompt":16,"max_input_tokens":16384,"max_output_tokens":2048,"max_tokens":2048,"max_video_length":2,"max_videos_per_prompt":1,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-1.0-pro-vision-001":{"mode":"chat","base_model":"gemini-1.0-pro-vision","deprecation_date":"2025-04-09","max_images_per_prompt":16,"max_input_tokens":16384,"max_output_tokens":2048,"max_tokens":2048,"max_video_length":2,"max_videos_per_prompt":1,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.0-ultra":{"mode":"chat","base_model":"gemini-1.0-ultra","max_input_tokens":8192,"max_output_tokens":2048,"max_tokens":2048,"source":"As of Jun, 2024. There is no available doc on vertex ai pricing gemini-1.0-ultra-001. Using gemini-1.0-pro pricing. Got max_tokens info here: https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-1.0-ultra-001":{"mode":"chat","base_model":"gemini-1.0-ultra","max_input_tokens":8192,"max_output_tokens":2048,"max_tokens":2048,"source":"As of Jun, 2024. There is no available doc on vertex ai pricing gemini-1.0-ultra-001. Using gemini-1.0-pro pricing. Got max_tokens info here: https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-1.5-flash":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1000000,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-flash-001":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-05-24","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1000000,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-flash-002":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-09-24","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-1.5-flash","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-flash-exp-0827":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1000000,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-flash-preview-0514":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1000000,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-pro":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-29","max_input_tokens":2097152,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-pro-001":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-05-24","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-pro-002":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-24","max_input_tokens":2097152,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-1.5-pro","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-pro-preview-0215":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-29","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-pro-preview-0409":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-29","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-1.5-pro-preview-0514":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-29","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-2.0-flash":{"mode":"chat","base_model":"gemini-2.0-flash","deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/pricing#2_0flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-2.0-flash-001":{"mode":"chat","base_model":"gemini-2.0-flash","deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini-2.0-flash-exp":{"mode":"chat","base_model":"gemini-2.0-flash","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.0-flash-lite":{"mode":"chat","base_model":"gemini-2.0-flash-lite","deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":50,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-2.0-flash-lite-001":{"mode":"chat","base_model":"gemini-2.0-flash-lite","deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":50,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini-2.0-flash-live-preview-04-09":{"mode":"chat","base_model":"gemini-2.0-flash-live","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10,"source":"https://cloud.google.com/vertex-ai/docs/generative-ai/model-reference/gemini#gemini-2-0-flash-live-preview-04-09","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_output":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.0-flash-preview-image-generation":{"mode":"chat","base_model":"gemini-2.0-flash-image-generation","deprecation_date":"2025-11-14","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/pricing#2_0flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-2.0-flash-thinking-exp":{"mode":"chat","base_model":"gemini-2.0-flash-thinking","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-2.0-flash-thinking-exp-01-21":{"mode":"chat","base_model":"gemini-2.0-flash-thinking","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65536,"max_pdf_size_mb":30,"max_tokens":65536,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":false,"supports_function_calling":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini-2.0-pro-exp-02-05":{"mode":"chat","base_model":"gemini-2.0-pro","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":2097152,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-flash":{"mode":"chat","base_model":"gemini-2.5-flash","deprecation_date":"2026-10-20","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"supports_image_size":false,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":false,"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":24576},"supports_response_schema_with_tools":false},"gemini-2.5-flash-image":{"mode":"image_generation","base_model":"gemini-2.5-flash-image","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"rpm":100000,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-flash-image","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":false,"tpm":8000000,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-flash-image-preview":{"mode":"image_generation","base_model":"gemini-2.5-flash-image","deprecation_date":"2026-01-15","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":100000,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":8000000,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-3-pro-image-preview":{"mode":"image_generation","base_model":"gemini-3-pro-image","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai-language-models","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"service_tiers":["flex"]},"deep-research-pro-preview-12-2025":{"mode":"image_generation","base_model":"deep-research-pro","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-flash-lite":{"mode":"chat","base_model":"gemini-2.5-flash-lite","deprecation_date":"2026-10-20","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"supports_image_size":false,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":false,"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":512,"max":24576},"supports_response_schema_with_tools":false},"gemini-2.5-flash-lite-preview-09-2025":{"mode":"chat","base_model":"gemini-2.5-flash-lite","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://developers.googleblog.com/en/continuing-to-bring-you-our-latest-models-with-an-improved-gemini-2-5-flash-and-flash-lite-release/","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"supports_image_size":false,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-flash-preview-09-2025":{"mode":"chat","base_model":"gemini-2.5-flash","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://developers.googleblog.com/en/continuing-to-bring-you-our-latest-models-with-an-improved-gemini-2-5-flash-and-flash-lite-release/","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"supports_image_size":false,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-live-2.5-flash-preview-native-audio-09-2025":{"mode":"chat","base_model":"gemini-live-2.5-flash-native-audio","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"gemini_native_audio":true,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-live-2.5-flash-preview-native-audio-09-2025":{"mode":"chat","base_model":"gemini-live-2.5-flash-native-audio","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":100000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":8000000,"gemini_native_audio":true,"provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-flash-lite-preview-06-17":{"mode":"chat","base_model":"gemini-2.5-flash-lite","deprecation_date":"2025-11-18","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini-2.5-flash-preview-04-17":{"mode":"chat","base_model":"gemini-2.5-flash","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini-2.5-flash-preview-05-20":{"mode":"chat","base_model":"gemini-2.5-flash","deprecation_date":"2025-11-18","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini-2.5-pro":{"mode":"chat","base_model":"gemini-2.5-pro","deprecation_date":"2026-10-20","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"prompt_cache_min_tokens":2048,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":false,"supports_reasoning_disable":false,"supports_none_reasoning_effort":false,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":128,"max":32768},"supports_response_schema_with_tools":false},"gemini-3-pro-preview":{"mode":"chat","base_model":"gemini-3-pro","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"provider":"vertex_ai-language-models","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call. Supports Google Search, File Search, Code Execution, URL Context, and Function Calling.","type":"select"},{"id":"media_resolution","label":"Media Resolution","helpText":"Higher resolutions may provide better understanding but use more tokens.","type":"select","default":"media_resolution_medium","options":[{"label":"Low","value":"media_resolution_low"},{"label":"Medium","value":"media_resolution_medium"},{"label":"High","value":"media_resolution_high"}]},{"id":"thinking_level","label":"Thinking Level","helpText":"Set the thinking level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"temperature","label":"Temperature","helpText":"For Gemini 3, best results at default 1.0. Lower values may impact reasoning.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Output length","helpText":"Maximum number of tokens in response","type":"number","default":32768,"range":{"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":5,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_multimodal_tool_output":true,"supports_response_schema_with_tools":true},"vertex_ai/gemini-3-pro-preview":{"mode":"chat","base_model":"gemini-3-pro","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_multimodal_tool_output":true,"supports_response_schema_with_tools":true},"vertex_ai/gemini-3-flash-preview":{"mode":"chat","base_model":"gemini-3-flash","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"supports_audio_input":true,"supports_video_input":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"service_tiers":["priority","flex"],"supports_multimodal_tool_output":true,"supports_response_schema_with_tools":true},"gemini-2.5-pro-exp-03-25":{"mode":"chat","base_model":"gemini-2.5-pro","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini-2.5-pro-preview-03-25":{"mode":"chat","base_model":"gemini-2.5-pro","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini-2.5-pro-preview-05-06":{"mode":"chat","base_model":"gemini-2.5-pro","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supported_regions":["global"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini-2.5-pro-preview-06-05":{"mode":"chat","base_model":"gemini-2.5-pro","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-pro-preview-tts":{"mode":"chat","base_model":"gemini-2.5-pro-tts","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-pro-preview","supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-robotics-er-1.5-preview":{"mode":"chat","base_model":"gemini-robotics-er-1.5","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-robotics-er-1-5-preview","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","video","audio"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-robotics-er-1.5-preview":{"mode":"chat","base_model":"gemini-robotics-er-1.5","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-robotics-er-1-5-preview","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","video","audio"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"rpm":10,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-computer-use-preview-10-2025":{"mode":"chat","base_model":"gemini-2.5-computer-use","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/computer-use","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","max_images_per_prompt":3000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-embedding-001":{"mode":"embedding","base_model":"gemini-embedding-001","deprecation_date":"2028-05-20","max_input_tokens":2048,"max_tokens":2048,"output_vector_size":3072,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models","provider":"vertex_ai"},"gemini-flash-experimental":{"mode":"chat","base_model":"gemini-flash","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/multimodal/gemini-experimental","supports_function_calling":false,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-pro":{"mode":"chat","base_model":"gemini-pro","max_input_tokens":32760,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-pro-experimental":{"mode":"chat","base_model":"gemini-pro","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/multimodal/gemini-experimental","supports_function_calling":false,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-pro-vision":{"mode":"chat","base_model":"gemini-pro-vision","max_images_per_prompt":16,"max_input_tokens":16384,"max_output_tokens":2048,"max_tokens":2048,"max_video_length":2,"max_videos_per_prompt":1,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-embedding-001":{"mode":"embedding","base_model":"gemini-embedding-001","deprecation_date":"2028-05-14","max_input_tokens":2048,"max_tokens":2048,"output_vector_size":3072,"rpm":10000,"source":"https://ai.google.dev/gemini-api/docs/embeddings#model-versions","tpm":10000000,"provider":"gemini"},"gemini/gemini-1.5-flash":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":2000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-flash-001":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-05-24","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":2000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-flash-002":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-09-24","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":2000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-flash-8b":{"mode":"chat","base_model":"gemini-1.5-flash-8b","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":4000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-flash-8b-exp-0827":{"mode":"chat","base_model":"gemini-1.5-flash-8b","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1000000,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":4000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-flash-8b-exp-0924":{"mode":"chat","base_model":"gemini-1.5-flash-8b","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":4000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-flash-exp-0827":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":2000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-flash-latest":{"mode":"chat","base_model":"gemini-1.5-flash","deprecation_date":"2025-09-29","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":2000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-pro":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-29","max_input_tokens":2097152,"max_output_tokens":8192,"max_tokens":8192,"rpm":1000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-pro-001":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-05-24","max_input_tokens":2097152,"max_output_tokens":8192,"max_tokens":8192,"rpm":1000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-pro-002":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-24","max_input_tokens":2097152,"max_output_tokens":8192,"max_tokens":8192,"rpm":1000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-pro-exp-0801":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-29","max_input_tokens":2097152,"max_output_tokens":8192,"max_tokens":8192,"rpm":1000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-pro-exp-0827":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-29","max_input_tokens":2097152,"max_output_tokens":8192,"max_tokens":8192,"rpm":1000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-1.5-pro-latest":{"mode":"chat","base_model":"gemini-1.5-pro","deprecation_date":"2025-09-29","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"rpm":1000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-2.0-flash":{"mode":"chat","base_model":"gemini-2.0-flash","provider":"gemini","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"is_deprecated":true,"deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10000,"source":"https://ai.google.dev/pricing#2_0flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_input":true,"supports_audio_output":true,"supports_url_context":true,"supports_web_search":true,"tpm":10000000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini/gemini-2.0-flash-001":{"mode":"chat","base_model":"gemini-2.0-flash","deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10000,"source":"https://ai.google.dev/pricing#2_0flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":false,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":10000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini/gemini-2.0-flash-exp":{"mode":"chat","base_model":"gemini-2.0-flash","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini/gemini-2.0-flash-lite":{"mode":"chat","base_model":"gemini-2.0-flash-lite","provider":"gemini","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":1064000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"is_deprecated":true,"deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":50,"max_video_length":1,"max_videos_per_prompt":10,"rpm":4000,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.0-flash-lite","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":true,"supports_web_search":true,"tpm":4000000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini/gemini-2.0-flash-lite-preview-02-05":{"mode":"chat","base_model":"gemini-2.0-flash-lite","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":60000,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash-lite","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":10000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini/gemini-2.0-flash-live-001":{"mode":"chat","base_model":"gemini-2.0-flash-live","deprecation_date":"2025-12-09","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2-0-flash-live-001","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_output":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-2.0-flash-preview-image-generation":{"mode":"chat","base_model":"gemini-2.0-flash-image-generation","deprecation_date":"2025-11-14","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10000,"source":"https://ai.google.dev/pricing#2_0flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":10000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-2.0-flash-thinking-exp":{"mode":"chat","base_model":"gemini-2.0-flash-thinking","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65536,"max_pdf_size_mb":30,"max_tokens":65536,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-2.0-flash-thinking-exp-01-21":{"mode":"chat","base_model":"gemini-2.0-flash-thinking","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65536,"max_pdf_size_mb":30,"max_tokens":65536,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#gemini-2.0-flash","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini/gemini-2.0-pro-exp-02-05":{"mode":"chat","base_model":"gemini-2.0-pro","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":2097152,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"rpm":2,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"tpm":1000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.5-flash":{"mode":"chat","base_model":"gemini-2.5-flash","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":100000,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":8000000,"supports_audio_input":true,"supports_image_size":false,"provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini/gemini-2.5-flash-image":{"mode":"image_generation","base_model":"gemini-2.5-flash-image","deprecation_date":"2026-10-02","supports_reasoning":false,"max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"rpm":100000,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-flash-image","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":8000000,"supports_audio_input":false,"supports_image_size":false,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.5-flash-image-preview":{"mode":"image_generation","base_model":"gemini-2.5-flash-image","deprecation_date":"2026-01-15","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":100000,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":8000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/gemini-3-pro-image-preview":{"mode":"image_generation","base_model":"gemini-3-pro-image","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"rpm":1000,"tpm":4000000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/deep-research-pro-preview-12-2025":{"mode":"image_generation","base_model":"deep-research-pro","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"rpm":1000,"tpm":4000000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.5-flash-lite":{"mode":"chat","base_model":"gemini-2.5-flash-lite","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":15,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-lite","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"supports_audio_input":true,"supports_image_size":false,"provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini/gemini-2.5-flash-lite-preview-09-2025":{"mode":"chat","base_model":"gemini-2.5-flash-lite","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":15,"source":"https://developers.googleblog.com/en/continuing-to-bring-you-our-latest-models-with-an-improved-gemini-2-5-flash-and-flash-lite-release/","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini/gemini-2.5-flash-preview-09-2025":{"mode":"chat","base_model":"gemini-2.5-flash","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":15,"source":"https://developers.googleblog.com/en/continuing-to-bring-you-our-latest-models-with-an-improved-gemini-2-5-flash-and-flash-lite-release/","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini/gemini-flash-latest":{"mode":"chat","base_model":"gemini-flash","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":15,"source":"https://developers.googleblog.com/en/continuing-to-bring-you-our-latest-models-with-an-improved-gemini-2-5-flash-and-flash-lite-release/","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"prompt_cache_min_tokens":4096,"supports_audio_input":true,"supports_native_streaming":true,"supports_video_input":true,"web_search_billing_unit":"per_query","provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-flash-lite-latest":{"mode":"chat","base_model":"gemini-flash-lite","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":15,"source":"https://developers.googleblog.com/en/continuing-to-bring-you-our-latest-models-with-an-improved-gemini-2-5-flash-and-flash-lite-release/","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"supports_audio_input":true,"supports_native_streaming":true,"supports_video_input":true,"web_search_billing_unit":"per_query","provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.5-flash-lite-preview-06-17":{"mode":"chat","base_model":"gemini-2.5-flash-lite","deprecation_date":"2025-11-18","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":15,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-lite","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini/gemini-2.5-flash-preview-04-17":{"mode":"chat","base_model":"gemini-2.5-flash","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}]},"gemini/gemini-2.5-flash-preview-05-20":{"mode":"chat","base_model":"gemini-2.5-flash","deprecation_date":"2025-11-18","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini/gemini-2.5-flash-preview-tts":{"mode":"audio_speech","base_model":"gemini-2.5-flash-tts","max_input_tokens":8192,"max_output_tokens":16384,"max_tokens":16384,"source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/audio/speech"],"tpm":4000000,"rpm":10,"supports_audio_input":false,"supports_function_calling":false,"supports_response_schema":false,"supports_web_search":false,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.5-pro":{"mode":"chat","base_model":"gemini-2.5-pro","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":2000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"prompt_cache_min_tokens":2048,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"tpm":800000,"provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"gemini/gemini-2.5-computer-use-preview-10-2025":{"mode":"chat","base_model":"gemini-2.5-computer-use","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/computer-use","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":800000,"provider":"gemini","max_images_per_prompt":3000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-3-pro-preview":{"mode":"chat","base_model":"gemini-3-pro","provider":"gemini","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"rpm":2000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_reasoning":true,"supports_video_input":true,"supports_web_search":true,"tpm":800000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call. Supports Google Search, File Search, Code Execution, URL Context, and Function Calling.","type":"select"},{"id":"media_resolution","label":"Media Resolution","helpText":"Higher resolutions may provide better understanding but use more tokens.","type":"select","default":"media_resolution_medium","options":[{"label":"Low","value":"media_resolution_low"},{"label":"Medium","value":"media_resolution_medium"},{"label":"High","value":"media_resolution_high"}]},{"id":"thinking_level","label":"Thinking Level","helpText":"Set the thinking level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"temperature","label":"Temperature","helpText":"For Gemini 3, best results at default 1.0. Lower values may impact reasoning.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Output length","helpText":"Maximum number of tokens in response","type":"number","default":32768,"range":{"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":5,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_multimodal_tool_output":true,"supports_response_schema_with_tools":true},"gemini/gemini-3-flash-preview":{"mode":"chat","base_model":"gemini-3-flash","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":2000,"source":"https://ai.google.dev/pricing/gemini-3","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"tpm":800000,"web_search_billing_unit":"per_query","supports_audio_input":true,"provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-3-flash-preview":{"mode":"chat","base_model":"gemini-3-flash","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"source":"https://ai.google.dev/pricing/gemini-3","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"provider":"vertex_ai","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["minimal","low","medium","high"],"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini/gemini-2.5-pro-exp-03-25":{"mode":"chat","base_model":"gemini-2.5-pro","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":5,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"gemini/gemini-2.5-pro-preview-03-25":{"mode":"chat","base_model":"gemini-2.5-pro","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10000,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-pro-preview","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":10000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"is_deprecated":true},"gemini/gemini-2.5-pro-preview-05-06":{"mode":"chat","base_model":"gemini-2.5-pro","deprecation_date":"2025-12-02","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10000,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-pro-preview","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":10000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini/gemini-2.5-pro-preview-06-05":{"mode":"chat","base_model":"gemini-2.5-pro","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":10000,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-pro-preview","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":10000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65535}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"gemini/gemini-2.5-pro-preview-tts":{"mode":"chat","base_model":"gemini-2.5-pro-tts","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":10000,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.5-pro-preview","supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_output":false,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":10000000,"supports_audio_input":false,"provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-exp-1114":{"mode":"chat","base_model":"gemini","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"metadata":{"notes":"Rate limits not documented for gemini-exp-1114. Assuming same as gemini-1.5-pro.","supports_tool_choice":true},"rpm":1000,"source":"https://ai.google.dev/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"tpm":4000000,"provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-exp-1206":{"mode":"chat","base_model":"gemini","max_input_tokens":2097152,"max_output_tokens":8192,"max_tokens":8192,"rpm":1000,"source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":4000000,"provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"metadata":{"notes":"Rate limits not documented for gemini-exp-1206. Assuming same as gemini-1.5-pro.","supports_tool_choice":true},"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-gemma-2-27b-it":{"mode":"chat","base_model":"gemma-2-27b-it","max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"tpm":250000,"rpm":10,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-gemma-2-9b-it":{"mode":"chat","base_model":"gemma-2-9b-it","max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"tpm":250000,"rpm":10,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-pro":{"mode":"chat","base_model":"gemini-pro","max_input_tokens":32760,"max_output_tokens":8192,"max_tokens":8192,"rpd":30000,"rpm":360,"source":"https://ai.google.dev/gemini-api/docs/models/gemini","supports_function_calling":true,"supports_tool_choice":true,"tpm":120000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-pro-vision":{"mode":"chat","base_model":"gemini-pro-vision","max_input_tokens":30720,"max_output_tokens":2048,"max_tokens":2048,"rpd":30000,"rpm":360,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"tpm":120000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemma-3-27b-it":{"mode":"chat","base_model":"gemma-3-27b-it","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aistudio.google.com","supports_audio_output":false,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/imagen-3.0-fast-generate-001":{"mode":"image_generation","base_model":"imagen-3.0-fast-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/imagen-3.0-generate-001":{"mode":"image_generation","base_model":"imagen-3.0-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/imagen-3.0-generate-002":{"mode":"image_generation","base_model":"imagen-3.0-generate","deprecation_date":"2025-11-10","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/imagen-4.0-fast-generate-001":{"mode":"image_generation","base_model":"imagen-4.0-fast-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/imagen-4.0-generate-001":{"mode":"image_generation","base_model":"imagen-4.0-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/imagen-4.0-ultra-generate-001":{"mode":"image_generation","base_model":"imagen-4.0-ultra-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/learnlm-1.5-pro-experimental":{"mode":"chat","base_model":"learnlm-1.5-pro","max_input_tokens":32767,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aistudio.google.com","supports_audio_output":false,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/veo-2.0-generate-001":{"mode":"video_generation","base_model":"veo-2.0-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/veo-3.0-fast-generate-preview":{"mode":"video_generation","base_model":"veo-3.0-fast-generate","deprecation_date":"2025-11-12","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/veo-3.0-generate-preview":{"mode":"video_generation","base_model":"veo-3.0-generate","deprecation_date":"2025-11-12","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"gemini/veo-3.1-fast-generate-preview":{"mode":"video_generation","base_model":"veo-3.1-fast-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/veo-3.1-generate-preview":{"mode":"video_generation","base_model":"veo-3.1-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/veo-3.1-fast-generate-001":{"mode":"video_generation","base_model":"veo-3.1-fast-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/veo-3.1-generate-001":{"mode":"video_generation","base_model":"veo-3.1-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/claude-haiku-4.5":{"mode":"chat","base_model":"claude-haiku-4-5","max_input_tokens":128000,"max_output_tokens":16000,"max_tokens":16000,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/claude-opus-4.5":{"mode":"chat","base_model":"claude-opus-4-5","max_input_tokens":128000,"max_output_tokens":16000,"max_tokens":16000,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_output_config":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/claude-opus-41":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":80000,"max_output_tokens":16000,"max_tokens":16000,"supported_endpoints":["/v1/chat/completions"],"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/claude-sonnet-4":{"mode":"chat","base_model":"claude-sonnet-4","max_input_tokens":128000,"max_output_tokens":16000,"max_tokens":16000,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/claude-sonnet-4.5":{"mode":"chat","base_model":"claude-sonnet-4-5","max_input_tokens":128000,"max_output_tokens":16000,"max_tokens":16000,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gemini-2.5-pro":{"mode":"chat","base_model":"gemini-2.5-pro","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gemini-3-pro-preview":{"mode":"chat","base_model":"gemini-3-pro","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-3.5-turbo":{"mode":"chat","base_model":"gpt-3.5-turbo","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-3.5-turbo-0613":{"mode":"chat","base_model":"gpt-3.5-turbo","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4":{"mode":"chat","base_model":"gpt-4","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4-0613":{"mode":"chat","base_model":"gpt-4","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4-o-preview":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":64000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4.1-2025-04-14":{"mode":"chat","base_model":"gpt-4.1","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-41-copilot":{"mode":"completion","base_model":"gpt-41-copilot","provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4o":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":64000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4o-2024-05-13":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":64000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4o-2024-08-06":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":64000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4o-2024-11-20":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":64000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4o-mini":{"mode":"chat","base_model":"gpt-4o-mini","max_input_tokens":64000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-4o-mini-2024-07-18":{"mode":"chat","base_model":"gpt-4o-mini","max_input_tokens":64000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-5":{"mode":"chat","base_model":"gpt-5","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-5.1-codex-max":{"mode":"responses","base_model":"gpt-5.1-codex-max","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/gpt-5.2":{"mode":"chat","base_model":"gpt-5.2","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"github_copilot/text-embedding-3-small":{"mode":"embedding","base_model":"text-embedding-3-small","max_input_tokens":8191,"max_tokens":8191,"provider":"github_copilot"},"github_copilot/text-embedding-3-small-inference":{"mode":"embedding","base_model":"text-embedding-3-small-inference","max_input_tokens":8191,"max_tokens":8191,"provider":"github_copilot"},"github_copilot/text-embedding-ada-002":{"mode":"embedding","base_model":"text-embedding-ada-002","max_input_tokens":8191,"max_tokens":8191,"provider":"github_copilot"},"chatgpt/gpt-5.2-codex":{"mode":"responses","base_model":"gpt-5.2-codex","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chatgpt/gpt-5.2":{"mode":"responses","base_model":"gpt-5.2","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chatgpt/gpt-5.1-codex-max":{"mode":"responses","base_model":"gpt-5.1-codex-max","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chatgpt/gpt-5.1-codex-mini":{"mode":"responses","base_model":"gpt-5.1-codex-mini","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gigachat/GigaChat-2-Lite":{"mode":"chat","base_model":"gigachat-2-lite","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"provider":"gigachat","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gigachat/GigaChat-2-Max":{"mode":"chat","base_model":"gigachat-2-max","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_vision":true,"provider":"gigachat","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gigachat/GigaChat-2-Pro":{"mode":"chat","base_model":"gigachat-2-pro","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_vision":true,"provider":"gigachat","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gigachat/Embeddings":{"mode":"embedding","base_model":"embeddings","max_input_tokens":512,"max_tokens":512,"output_vector_size":1024,"provider":"gigachat"},"gigachat/Embeddings-2":{"mode":"embedding","base_model":"embeddings-2","max_input_tokens":512,"max_tokens":512,"output_vector_size":1024,"provider":"gigachat"},"gigachat/EmbeddingsGigaR":{"mode":"embedding","base_model":"embeddingsgigar","max_input_tokens":4096,"max_tokens":4096,"output_vector_size":2560,"provider":"gigachat"},"gmi/anthropic/claude-opus-4.5":{"mode":"chat","base_model":"claude-opus-4-5","max_input_tokens":409600,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_vision":true,"supports_output_config":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/anthropic/claude-sonnet-4.5":{"mode":"chat","base_model":"claude-sonnet-4-5","max_input_tokens":409600,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_vision":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/anthropic/claude-sonnet-4":{"mode":"chat","base_model":"claude-sonnet-4","max_input_tokens":409600,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_vision":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/anthropic/claude-opus-4":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":409600,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_vision":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/openai/gpt-5.2":{"mode":"chat","base_model":"gpt-5.2","max_input_tokens":409600,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/openai/gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","max_input_tokens":409600,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/openai/gpt-5":{"mode":"chat","base_model":"gpt-5","max_input_tokens":409600,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/openai/gpt-4o":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_vision":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/openai/gpt-4o-mini":{"mode":"chat","base_model":"gpt-4o-mini","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_vision":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/deepseek-ai/DeepSeek-V3.2":{"mode":"chat","base_model":"deepseek-v3.2","max_input_tokens":163840,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/deepseek-ai/DeepSeek-V3-0324":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":163840,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/google/gemini-3-pro-preview":{"mode":"chat","base_model":"gemini-3-pro","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_vision":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/google/gemini-3-flash-preview":{"mode":"chat","base_model":"gemini-3-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_vision":true,"supports_system_messages":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/moonshotai/Kimi-K2-Thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":16384,"max_tokens":16384,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/MiniMaxAI/MiniMax-M2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196608,"max_output_tokens":16384,"max_tokens":16384,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/Qwen/Qwen3-VL-235B-A22B-Instruct-FP8":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct-fp8","max_input_tokens":262144,"max_output_tokens":16384,"max_tokens":16384,"supports_vision":true,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gmi/zai-org/GLM-4.7-FP8":{"mode":"chat","base_model":"glm-4.7-fp8","max_input_tokens":202752,"max_output_tokens":16384,"max_tokens":16384,"provider":"gmi","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"google.gemma-3-12b-it":{"mode":"chat","base_model":"gemma-3-12b-it","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"google.gemma-3-27b-it":{"mode":"chat","base_model":"gemma-3-27b-it","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"google.gemma-3-4b-it":{"mode":"chat","base_model":"gemma-3-4b-it","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_system_messages":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"google_pse/search":{"mode":"search","base_model":"search","provider":"google_pse","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"global.anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"global.anthropic.claude-sonnet-4-20250514-v1:0":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-10-14","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"bedrock_converse_supports_strict_tools":false,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"global.anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"global.amazon.nova-2-lite-v1:0":{"mode":"chat","base_model":"nova-2-lite","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-3.5-turbo":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2026-10-23","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_parallel_function_calling":true,"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-3.5-turbo-0125":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2026-10-23","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-3.5-turbo-0301":{"mode":"chat","base_model":"gpt-3.5-turbo","max_input_tokens":4097,"max_output_tokens":4096,"max_tokens":4096,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-3.5-turbo-0613":{"mode":"chat","base_model":"gpt-3.5-turbo","max_input_tokens":4097,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-3.5-turbo-1106":{"mode":"chat","base_model":"gpt-3.5-turbo","deprecation_date":"2026-09-28","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-3.5-turbo-16k":{"mode":"chat","base_model":"gpt-3.5-turbo-16k","deprecation_date":"2026-10-23","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-3.5-turbo-16k-0613":{"mode":"chat","base_model":"gpt-3.5-turbo-16k","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-3.5-turbo-instruct":{"mode":"completion","base_model":"gpt-3.5-turbo-instruct","deprecation_date":"2026-09-28","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","provider":"text-completion-openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":false,"supports_tool_choice":false,"supports_parallel_function_calling":false,"supports_reasoning":false,"supports_response_schema":false,"supports_prompt_caching":false,"supports_web_search":false,"supports_service_tier":false,"supports_assistant_prefill":false,"supports_system_messages":false},"gpt-3.5-turbo-instruct-0914":{"mode":"completion","base_model":"gpt-3.5-turbo-instruct","max_input_tokens":8192,"max_output_tokens":4097,"max_tokens":4097,"provider":"text-completion-openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4097}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":false,"supports_tool_choice":false,"supports_parallel_function_calling":false,"supports_reasoning":false,"supports_response_schema":false,"supports_prompt_caching":false,"supports_web_search":false,"supports_service_tier":false,"supports_assistant_prefill":false,"supports_system_messages":false},"gpt-4":{"mode":"chat","base_model":"gpt-4","deprecation_date":"2026-10-23","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_parallel_function_calling":true,"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-4-0125-preview":{"mode":"chat","base_model":"gpt-4","deprecation_date":"2026-03-26","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"is_deprecated":true},"gpt-4-0314":{"mode":"chat","base_model":"gpt-4","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4-0613":{"mode":"chat","base_model":"gpt-4","deprecation_date":"2025-06-06","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","is_deprecated":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_parallel_function_calling":true,"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-4-1106-preview":{"mode":"chat","base_model":"gpt-4","deprecation_date":"2026-03-26","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logitBias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"topLogprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presencePenalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"gpt-4-1106-vision-preview":{"mode":"chat","base_model":"gpt-4-1106-vision","deprecation_date":"2024-12-06","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"is_deprecated":true},"gpt-4-32k":{"mode":"chat","base_model":"gpt-4-32k","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4-32k-0314":{"mode":"chat","base_model":"gpt-4-32k","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4-32k-0613":{"mode":"chat","base_model":"gpt-4-32k","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4-turbo":{"mode":"chat","base_model":"gpt-4-turbo","deprecation_date":"2026-10-23","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-4-turbo-2024-04-09":{"mode":"chat","base_model":"gpt-4-turbo","deprecation_date":"2026-10-23","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-4-turbo-preview":{"mode":"chat","base_model":"gpt-4-turbo","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4-vision-preview":{"mode":"chat","base_model":"gpt-4-vision","deprecation_date":"2024-12-06","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"is_deprecated":true},"gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_assistant_prefill":true},"gpt-4.1-2025-04-14":{"mode":"chat","base_model":"gpt-4.1","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_assistant_prefill":true},"gpt-4.1-mini":{"mode":"chat","base_model":"gpt-4.1-mini","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_assistant_prefill":true},"gpt-4.1-mini-2025-04-14":{"mode":"chat","base_model":"gpt-4.1-mini","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_assistant_prefill":true},"gpt-4.1-nano":{"mode":"chat","base_model":"gpt-4.1-nano","deprecation_date":"2026-10-23","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_web_search":false,"supports_assistant_prefill":true},"gpt-4.1-nano-2025-04-14":{"mode":"chat","base_model":"gpt-4.1-nano","deprecation_date":"2026-10-23","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_web_search":false,"supports_assistant_prefill":true},"gpt-4.5-preview":{"mode":"chat","base_model":"gpt-4.5","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"gpt-4.5-preview-2025-02-27":{"mode":"chat","base_model":"gpt-4.5","deprecation_date":"2025-07-14","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"is_deprecated":true},"gpt-4o":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_web_search":true,"supports_assistant_prefill":true},"gpt-4o-2024-05-13":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-10-23","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":true,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-4o-2024-08-06":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_web_search":true,"supports_assistant_prefill":true},"gpt-4o-2024-11-20":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_web_search":true,"supports_assistant_prefill":true},"gpt-4o-audio-preview":{"mode":"chat","base_model":"gpt-4o-audio","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"gpt-4o-audio-preview-2024-10-01":{"mode":"chat","base_model":"gpt-4o-audio","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"gpt-4o-audio-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-audio","deprecation_date":"2027-01-20","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"gpt-4o-audio-preview-2025-06-03":{"mode":"chat","base_model":"gpt-4o-audio","deprecation_date":"2027-01-20","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"gpt-audio":{"mode":"chat","base_model":"gpt-audio","deprecation_date":"2027-01-20","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/realtime","/v1/batch"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-audio-2025-08-28":{"mode":"chat","base_model":"gpt-audio","deprecation_date":"2027-01-20","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/realtime","/v1/batch"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-audio-mini":{"mode":"chat","base_model":"gpt-audio-mini","deprecation_date":"2027-01-20","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/realtime","/v1/batch"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-audio-mini-2025-10-06":{"mode":"chat","base_model":"gpt-audio-mini","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/realtime","/v1/batch"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-audio-mini-2025-12-15":{"mode":"chat","base_model":"gpt-audio-mini","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/realtime","/v1/batch"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"deprecation_date":"2027-01-20","source":"https://developers.openai.com/api/docs/pricing","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini":{"mode":"chat","base_model":"gpt-4o-mini","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_web_search":true,"supports_assistant_prefill":true},"gpt-4o-mini-2024-07-18":{"mode":"chat","base_model":"gpt-4o-mini","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"prediction","label":"Predicted Output","helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","type":"text"},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_web_search":true,"supports_assistant_prefill":true},"gpt-4o-mini-audio-preview":{"mode":"chat","base_model":"gpt-4o-mini-audio","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"gpt-4o-mini-audio-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-mini-audio","deprecation_date":"2027-01-20","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-realtime-preview":{"mode":"chat","base_model":"gpt-4o-mini-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-realtime-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-mini-realtime","deprecation_date":"2027-01-20","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-search-preview":{"mode":"chat","base_model":"gpt-4o-mini-search","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-search-preview-2025-03-11":{"mode":"chat","base_model":"gpt-4o-mini-search","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-transcribe":{"mode":"audio_transcription","base_model":"gpt-4o-mini-transcribe","max_input_tokens":16000,"max_output_tokens":2000,"supported_endpoints":["/v1/audio/transcriptions"],"deprecation_date":"2027-02-26","source":"https://developers.openai.com/api/docs/pricing","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-tts":{"mode":"audio_speech","base_model":"gpt-4o-mini-tts","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/audio/speech"],"supported_modalities":["text","audio"],"supported_output_modalities":["audio"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-realtime-preview":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-realtime-preview-2024-10-01":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-realtime-preview-2024-12-17":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-realtime-preview-2025-06-03":{"mode":"chat","base_model":"gpt-4o-realtime","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-search-preview":{"mode":"chat","base_model":"gpt-4o-search","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-search-preview-2025-03-11":{"mode":"chat","base_model":"gpt-4o-search","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-transcribe":{"mode":"audio_transcription","base_model":"gpt-4o-transcribe","max_input_tokens":16000,"max_output_tokens":2000,"supported_endpoints":["/v1/audio/transcriptions"],"deprecation_date":"2027-02-26","source":"https://developers.openai.com/api/docs/pricing","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","deprecation_date":"2026-12-01","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/images/generations"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","deprecation_date":"2026-12-01","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/images/generations"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1024-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1024-x-1536/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1536-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1024-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1024-x-1536/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1536-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1024-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1024-x-1536/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1536-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1024-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1024-x-1536/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1536-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1024-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1024-x-1536/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1536-x-1024/gpt-image-1.5":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1024-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1024-x-1536/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1536-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1024-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1024-x-1536/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1536-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1024-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1024-x-1536/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1536-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1024-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1024-x-1536/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1536-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1024-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1024-x-1536/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"1536-x-1024/gpt-image-1.5-2025-12-16":{"mode":"image_generation","base_model":"gpt-image-1.5","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-5":{"mode":"chat","base_model":"gpt-5","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["minimal","low","medium","high"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true},"gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://developers.openai.com/api/docs/pricing","supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.1-2025-11-13":{"mode":"chat","base_model":"gpt-5.1","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://developers.openai.com/api/docs/pricing","supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.1-chat-latest":{"mode":"chat","base_model":"gpt-5.1-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_native_streaming":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"gpt-5.2":{"mode":"chat","base_model":"gpt-5.2","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://developers.openai.com/api/docs/pricing","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.2-2025-12-11":{"mode":"chat","base_model":"gpt-5.2","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://developers.openai.com/api/docs/pricing","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.2-chat-latest":{"mode":"chat","base_model":"gpt-5.2-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-5.2-pro":{"mode":"responses","base_model":"gpt-5.2-pro","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["medium","high","xhigh"],"supports_service_tier":true},"gpt-5.2-pro-2025-12-11":{"mode":"responses","base_model":"gpt-5.2-pro","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["medium","high","xhigh"],"supports_service_tier":true},"gpt-5-pro":{"mode":"responses","base_model":"gpt-5-pro","max_input_tokens":128000,"max_output_tokens":272000,"max_tokens":272000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":false,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":272000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["high"],"supports_service_tier":true},"gpt-5-pro-2025-10-06":{"mode":"responses","base_model":"gpt-5-pro","deprecation_date":"2026-12-11","max_input_tokens":128000,"max_output_tokens":272000,"max_tokens":272000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":false,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":272000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["high"],"supports_service_tier":true},"gpt-5-2025-08-07":{"mode":"chat","base_model":"gpt-5","deprecation_date":"2026-12-11","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["minimal","low","medium","high"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true},"gpt-5-chat":{"mode":"chat","base_model":"gpt-5-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_native_streaming":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"gpt-5-chat-latest":{"mode":"chat","base_model":"gpt-5-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_native_streaming":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}]},"gpt-5-codex":{"mode":"responses","base_model":"gpt-5-codex","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-5.1-codex":{"mode":"responses","base_model":"gpt-5.1-codex","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-5.1-codex-max":{"mode":"responses","base_model":"gpt-5.1-codex-max","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-5.1-codex-mini":{"mode":"responses","base_model":"gpt-5.1-codex-mini","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-5.2-codex":{"mode":"responses","base_model":"gpt-5.2-codex","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["minimal","low","medium","high"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true},"gpt-5-mini-2025-08-07":{"mode":"chat","base_model":"gpt-5-mini","deprecation_date":"2026-12-11","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["minimal","low","medium","high"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true},"gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":true,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["minimal","low","medium","high"],"supports_service_tier":true,"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true},"gpt-5-nano-2025-08-07":{"mode":"chat","base_model":"gpt-5-nano","deprecation_date":"2026-12-11","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":true,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["minimal","low","medium","high"],"supports_service_tier":true,"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true},"gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","deprecation_date":"2026-10-23","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","deprecation_date":"2026-12-01","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-realtime":{"mode":"chat","base_model":"gpt-realtime","deprecation_date":"2027-01-20","max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"instructions","label":"Instructions","helpText":"The default system instructions (i.e. system message) prepended to model calls.","type":"text"},{"id":"voice","label":"Voice","helpText":"Pre-selected voice used when generating the audio","type":"select","default":"alloy","options":[{"label":"Alloy","value":"alloy"},{"label":"Ash","value":"ash"},{"label":"Ballad","value":"ballad"},{"label":"Coral","value":"coral"},{"label":"Echo","value":"echo"},{"label":"Sage","value":"sage"},{"label":"Shimmer","value":"shimmer"},{"label":"Verse","value":"verse"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0.6,"max":1.2,"step":0.01}},{"id":"max_response_output_tokens","label":"Max Response Output Tokens","helpText":"Maximum number of output tokens for a single assistant response, inclusive of tool calls.","type":"number","default":510},{"id":"input_audio_noise_reduction","label":"Input Audio Noise Reduction","helpText":"Noise reduction applied to audio input, helpful with VAD and model understanding.","type":"select","accesorKey":"type","options":[{"label":"None","value":"none"},{"label":"Near Field","value":"near_field"},{"label":"Far Field","value":"far_field"}]},{"id":"speed","label":"Speed","helpText":"The speed of the model's spoken response. ","type":"number","default":1,"range":{"min":0.25,"max":1.5,"step":0.05}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"gpt-realtime-mini":{"mode":"chat","base_model":"gpt-realtime-mini","deprecation_date":"2027-01-20","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-realtime-2025-08-28":{"mode":"chat","base_model":"gpt-realtime","deprecation_date":"2027-01-20","max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"instructions","label":"Instructions","helpText":"The default system instructions (i.e. system message) prepended to model calls.","type":"text"},{"id":"voice","label":"Voice","helpText":"Pre-selected voice used when generating the audio","type":"select","default":"alloy","options":[{"label":"Alloy","value":"alloy"},{"label":"Ash","value":"ash"},{"label":"Ballad","value":"ballad"},{"label":"Coral","value":"coral"},{"label":"Echo","value":"echo"},{"label":"Sage","value":"sage"},{"label":"Shimmer","value":"shimmer"},{"label":"Verse","value":"verse"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0.6,"max":1.2,"step":0.01}},{"id":"max_response_output_tokens","label":"Max Response Output Tokens","helpText":"Maximum number of output tokens for a single assistant response, inclusive of tool calls.","type":"number","default":510},{"id":"input_audio_noise_reduction","label":"Input Audio Noise Reduction","helpText":"Noise reduction applied to audio input, helpful with VAD and model understanding.","type":"select","accesorKey":"type","options":[{"label":"None","value":"none"},{"label":"Near Field","value":"near_field"},{"label":"Far Field","value":"far_field"}]},{"id":"speed","label":"Speed","helpText":"The speed of the model's spoken response. ","type":"number","default":1,"range":{"min":0.25,"max":1.5,"step":0.05}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"gradient_ai/alibaba-qwen3-32b":{"mode":"chat","base_model":"qwen3-32b","max_tokens":2048,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":131072,"max_output_tokens":40960,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":2048}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"gradient_ai/anthropic-claude-3-opus":{"mode":"chat","base_model":"claude-3-opus","max_tokens":1024,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":200000,"max_output_tokens":1024,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/anthropic-claude-3.5-haiku":{"mode":"chat","base_model":"claude-3-5-haiku","max_tokens":1024,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":200000,"max_output_tokens":1024,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/anthropic-claude-3.5-sonnet":{"mode":"chat","base_model":"claude-3-5-sonnet","max_tokens":1024,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":200000,"max_output_tokens":1024,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/anthropic-claude-3.7-sonnet":{"mode":"chat","base_model":"claude-3-7-sonnet","max_tokens":1024,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":200000,"max_output_tokens":1024,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/deepseek-r1-distill-llama-70b":{"mode":"chat","base_model":"deepseek-r1-distill-llama-70b","max_tokens":8000,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":32768,"max_output_tokens":8000,"provider":"gradient_ai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"gradient_ai/llama3-8b-instruct":{"mode":"chat","base_model":"llama-3-8b-instruct","max_tokens":512,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":8192,"max_output_tokens":512,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/llama3.3-70b-instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_tokens":2048,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":128000,"max_output_tokens":2048,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/mistral-nemo-instruct-2407":{"mode":"chat","base_model":"mistral-nemo-instruct","max_tokens":512,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":128000,"max_output_tokens":512,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/openai-gpt-4o":{"mode":"chat","base_model":"gpt-4o","max_tokens":16384,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":128000,"max_output_tokens":16384,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/openai-gpt-4o-mini":{"mode":"chat","base_model":"gpt-4o-mini","max_tokens":16384,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":128000,"max_output_tokens":16384,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/openai-o3":{"mode":"chat","base_model":"o3","max_tokens":100000,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":200000,"max_output_tokens":100000,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gradient_ai/openai-o3-mini":{"mode":"chat","base_model":"o3-mini","max_tokens":100000,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supports_tool_choice":false,"max_input_tokens":200000,"max_output_tokens":100000,"provider":"gradient_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lemonade/Qwen3-Coder-30B-A3B-Instruct-GGUF":{"mode":"chat","base_model":"qwen3-coder-30b-a3b-instruct","max_tokens":32768,"max_input_tokens":262144,"max_output_tokens":32768,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"lemonade","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lemonade/gpt-oss-20b-mxfp4-GGUF":{"mode":"chat","base_model":"gpt-oss-20b","max_tokens":32768,"max_input_tokens":131072,"max_output_tokens":32768,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"lemonade","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"lemonade/gpt-oss-120b-mxfp-GGUF":{"mode":"chat","base_model":"gpt-oss-120b","max_tokens":32768,"max_input_tokens":131072,"max_output_tokens":32768,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"lemonade","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"lemonade/Gemma-3-4b-it-GGUF":{"mode":"chat","base_model":"gemma-3-4b-it","max_tokens":8192,"max_input_tokens":128000,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"lemonade","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lemonade/Qwen3-4B-Instruct-2507-GGUF":{"mode":"chat","base_model":"qwen3-4b-instruct","max_tokens":32768,"max_input_tokens":262144,"max_output_tokens":32768,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"lemonade","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon-nova/nova-micro-v1":{"mode":"chat","base_model":"nova-micro-v1","max_input_tokens":128000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"amazon_nova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon-nova/nova-lite-v1":{"mode":"chat","base_model":"nova-lite-v1","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"provider":"amazon_nova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon-nova/nova-premier-v1":{"mode":"chat","base_model":"nova-premier-v1","max_input_tokens":1000000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_response_schema":true,"supports_vision":true,"provider":"amazon_nova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"amazon-nova/nova-pro-v1":{"mode":"chat","base_model":"nova-pro-v1","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"provider":"amazon_nova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"groq/llama-3.1-8b-instant":{"mode":"chat","base_model":"llama-3.1-8b-instant","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"groq/llama-3.3-70b-versatile":{"mode":"chat","base_model":"llama-3.3-70b-versatile","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"groq/gemma-7b-it":{"mode":"chat","base_model":"gemma-7b-it","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"groq/meta-llama/llama-guard-4-12b":{"mode":"chat","base_model":"llama-guard-4-12b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"groq/meta-llama/llama-4-maverick-17b-128e-instruct":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"groq","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}]},"groq/meta-llama/llama-4-scout-17b-16e-instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"groq","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}]},"groq/moonshotai/kimi-k2-instruct-0905":{"mode":"chat","base_model":"kimi-k2-instruct","max_input_tokens":262144,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"groq/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32766,"max_tokens":32766,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32766}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"groq/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"groq/playai-tts":{"mode":"audio_speech","base_model":"playai-tts","max_input_tokens":10000,"max_output_tokens":10000,"max_tokens":10000,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"groq/qwen/qwen3-32b":{"mode":"chat","base_model":"qwen3-32b","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":131000}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131000,"range":{"min":1,"max":131000}}]},"groq/whisper-large-v3":{"mode":"audio_transcription","base_model":"whisper-large-v3","provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"groq/whisper-large-v3-turbo":{"mode":"audio_transcription","base_model":"whisper-large-v3-turbo","provider":"groq","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hd/1024-x-1024/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hd/1024-x-1792/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hd/1792-x-1024/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"heroku/claude-3-5-haiku":{"mode":"chat","base_model":"claude-3-5-haiku","max_tokens":4096,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"max_input_tokens":200000,"max_output_tokens":8192,"provider":"heroku","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"heroku/claude-3-5-sonnet-latest":{"mode":"chat","base_model":"claude-3-5-sonnet","max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"max_input_tokens":200000,"max_output_tokens":8192,"provider":"heroku","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"heroku/claude-3-7-sonnet":{"mode":"chat","base_model":"claude-3-7-sonnet","max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"max_input_tokens":200000,"max_output_tokens":8192,"provider":"heroku","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"heroku/claude-4-sonnet":{"mode":"chat","base_model":"claude-sonnet-4","max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"max_input_tokens":200000,"max_output_tokens":8192,"provider":"heroku","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1024-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1024-x-1536/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"high/1536-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/NousResearch/Hermes-3-Llama-3.1-70B":{"mode":"chat","base_model":"hermes-3-llama-3.1-70b","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/Qwen/QwQ-32B":{"mode":"chat","base_model":"qwq-32b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"hyperbolic/Qwen/Qwen2.5-72B-Instruct":{"mode":"chat","base_model":"qwen2.5-72b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"hyperbolic/Qwen/Qwen2.5-Coder-32B-Instruct":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/Qwen/Qwen3-235B-A22B":{"mode":"chat","base_model":"qwen3-235b-a22b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"hyperbolic/deepseek-ai/DeepSeek-R1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/deepseek-ai/DeepSeek-R1-0528":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/deepseek-ai/DeepSeek-V3":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/deepseek-ai/DeepSeek-V3-0324":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/meta-llama/Llama-3.2-3B-Instruct":{"mode":"chat","base_model":"llama-3.2-3b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/meta-llama/Meta-Llama-3-70B-Instruct":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"hyperbolic/meta-llama/Meta-Llama-3.1-405B-Instruct":{"mode":"chat","base_model":"llama-3.1-405b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/meta-llama/Meta-Llama-3.1-70B-Instruct":{"mode":"chat","base_model":"llama-3.1-70b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/meta-llama/Meta-Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"hyperbolic/moonshotai/Kimi-K2-Instruct":{"mode":"chat","base_model":"kimi-k2-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"hyperbolic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"j2-light":{"mode":"completion","base_model":"j2-light","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"j2-mid":{"mode":"completion","base_model":"j2-mid","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"j2-ultra":{"mode":"completion","base_model":"j2-ultra","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-1.5":{"mode":"chat","base_model":"jamba-1.5","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-1.5-large":{"mode":"chat","base_model":"jamba-1.5-large","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-1.5-large@001":{"mode":"chat","base_model":"jamba-1.5-large","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-1.5-mini":{"mode":"chat","base_model":"jamba-1.5-mini","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-1.5-mini@001":{"mode":"chat","base_model":"jamba-1.5-mini","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-large-1.6":{"mode":"chat","base_model":"jamba-large-1.6","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-large-1.7":{"mode":"chat","base_model":"jamba-large-1.7","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-mini-1.6":{"mode":"chat","base_model":"jamba-mini-1.6","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jamba-mini-1.7":{"mode":"chat","base_model":"jamba-mini-1.7","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"ai21","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jina-reranker-v2-base-multilingual":{"mode":"rerank","base_model":"jina-reranker-v2-base-multilingual","max_input_tokens":1024,"max_output_tokens":1024,"max_tokens":1024,"source":"https://api.jina.ai/v1/models","provider":"jina_ai","max_document_chunks_per_query":2048,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"jp.anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"jp.anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"lambda_ai/deepseek-llama3.3-70b":{"mode":"chat","base_model":"deepseek-llama3.3-70b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/deepseek-r1-0528":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/deepseek-r1-671b":{"mode":"chat","base_model":"deepseek-r1-671b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"lambda_ai/deepseek-v3-0324":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/hermes3-405b":{"mode":"chat","base_model":"hermes3-405b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"lambda_ai/hermes3-70b":{"mode":"chat","base_model":"hermes3-70b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"lambda_ai/hermes3-8b":{"mode":"chat","base_model":"hermes3-8b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"lambda_ai/lfm-40b":{"mode":"chat","base_model":"lfm-40b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/lfm-7b":{"mode":"chat","base_model":"lfm-7b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/llama-4-maverick-17b-128e-instruct-fp8":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct-fp8","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/llama-4-scout-17b-16e-instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_input_tokens":16384,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"lambda_ai/llama3.1-405b-instruct-fp8":{"mode":"chat","base_model":"llama-3.1-405b-instruct-fp8","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/llama3.1-70b-instruct-fp8":{"mode":"chat","base_model":"llama-3.1-70b-instruct-fp8","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/llama3.1-8b-instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/llama3.1-nemotron-70b-instruct-fp8":{"mode":"chat","base_model":"llama-3.1-nemotron-70b-instruct-fp8","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/llama3.2-11b-vision-instruct":{"mode":"chat","base_model":"llama-3.2-11b-vision-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/llama3.2-3b-instruct":{"mode":"chat","base_model":"llama-3.2-3b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/llama3.3-70b-instruct-fp8":{"mode":"chat","base_model":"llama-3.3-70b-instruct-fp8","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/qwen25-coder-32b-instruct":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"lambda_ai/qwen3-32b-fp8":{"mode":"chat","base_model":"qwen3-32b-fp8","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"lambda_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1024-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1024-x-1536/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1536-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"luminous-base":{"mode":"completion","base_model":"luminous-base","max_tokens":2048,"provider":"aleph_alpha","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"luminous-base-control":{"mode":"chat","base_model":"luminous-base-control","max_tokens":2048,"provider":"aleph_alpha","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"luminous-extended":{"mode":"completion","base_model":"luminous-extended","max_tokens":2048,"provider":"aleph_alpha","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"luminous-extended-control":{"mode":"chat","base_model":"luminous-extended-control","max_tokens":2048,"provider":"aleph_alpha","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"luminous-supreme":{"mode":"completion","base_model":"luminous-supreme","max_tokens":2048,"provider":"aleph_alpha","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"luminous-supreme-control":{"mode":"chat","base_model":"luminous-supreme-control","max_tokens":2048,"provider":"aleph_alpha","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"max-x-max/50-steps/stability.stable-diffusion-xl-v0":{"mode":"image_generation","base_model":"stable-diffusion-xl-v0","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"max-x-max/max-steps/stability.stable-diffusion-xl-v0":{"mode":"image_generation","base_model":"stable-diffusion-xl-v0","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1024-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1024-x-1536/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1536-x-1024/gpt-image-1":{"mode":"image_generation","base_model":"gpt-image-1","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1024-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1024-x-1536/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"low/1536-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1024-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1024-x-1536/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medium/1536-x-1024/gpt-image-1-mini":{"mode":"image_generation","base_model":"gpt-image-1-mini","supported_endpoints":["/v1/images/generations"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medlm-large":{"mode":"chat","base_model":"medlm-large","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"medlm-medium":{"mode":"chat","base_model":"medlm-medium","max_input_tokens":32768,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta.llama2-13b-chat-v1":{"mode":"chat","base_model":"llama-2-13b-chat-v1","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta.llama2-70b-chat-v1":{"mode":"chat","base_model":"llama-2-70b-chat-v1","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta.llama3-1-405b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-1-405b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta.llama3-1-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-1-70b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}}]},"meta.llama3-1-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-1-8b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}}]},"meta.llama3-2-11b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-11b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"meta.llama3-2-1b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-1b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"meta.llama3-2-3b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-3b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"meta.llama3-2-90b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-90b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"meta.llama3-3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-3-70b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"bedrock","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}}]},"meta.llama4-maverick-17b-instruct-v1:0":{"mode":"chat","base_model":"llama-4-maverick-17b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_modalities":["text","image"],"supported_output_modalities":["text","code"],"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta.llama4-scout-17b-instruct-v1:0":{"mode":"chat","base_model":"llama-4-scout-17b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_modalities":["text","image"],"supported_output_modalities":["text","code"],"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta_llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_input_tokens":128000,"max_output_tokens":4028,"max_tokens":4028,"source":"https://llama.developer.meta.com/docs/models","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"meta_llama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":4028}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta_llama/Llama-3.3-8B-Instruct":{"mode":"chat","base_model":"llama-3.3-8b-instruct","max_input_tokens":128000,"max_output_tokens":4028,"max_tokens":4028,"source":"https://llama.developer.meta.com/docs/models","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"meta_llama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4028}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"meta_llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct-fp8","max_input_tokens":1000000,"max_output_tokens":4028,"max_tokens":4028,"source":"https://llama.developer.meta.com/docs/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"meta_llama","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":4028}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"min_p","label":"Min P","helpText":"A number between 0 and 1 that can be used as an alternative to top_p and top-k","type":"number","default":0,"range":{"min":0,"max":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}]},"meta_llama/Llama-4-Scout-17B-16E-Instruct-FP8":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct-fp8","max_input_tokens":10000000,"max_output_tokens":4028,"max_tokens":4028,"source":"https://llama.developer.meta.com/docs/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"meta_llama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4028}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax.minimax-m2":{"mode":"chat","base_model":"minimax-m2","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":196000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/speech-02-hd":{"mode":"audio_speech","base_model":"speech-02-hd","supported_endpoints":["/v1/audio/speech"],"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/speech-02-turbo":{"mode":"audio_speech","base_model":"speech-02-turbo","supported_endpoints":["/v1/audio/speech"],"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/speech-2.6-hd":{"mode":"audio_speech","base_model":"speech-2.6-hd","supported_endpoints":["/v1/audio/speech"],"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/speech-2.6-turbo":{"mode":"audio_speech","base_model":"speech-2.6-turbo","supported_endpoints":["/v1/audio/speech"],"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/MiniMax-M2.1":{"mode":"chat","base_model":"minimax-m2.1","supports_function_calling":true,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_system_messages":true,"max_input_tokens":1000000,"max_output_tokens":8192,"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/MiniMax-M2.1-lightning":{"mode":"chat","base_model":"minimax-m2.1-lightning","supports_function_calling":true,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_system_messages":true,"max_input_tokens":1000000,"max_output_tokens":8192,"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/MiniMax-M2.5":{"mode":"chat","base_model":"minimax-m2.5","supports_function_calling":true,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_system_messages":true,"max_input_tokens":1000000,"max_output_tokens":8192,"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/MiniMax-M2.5-lightning":{"mode":"chat","base_model":"minimax-m2.5-lightning","supports_function_calling":true,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_system_messages":true,"max_input_tokens":1000000,"max_output_tokens":8192,"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"minimax/MiniMax-M2":{"mode":"chat","base_model":"minimax-m2","supports_function_calling":true,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_system_messages":true,"max_input_tokens":200000,"max_output_tokens":8192,"provider":"minimax","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.magistral-small-2509":{"mode":"chat","base_model":"magistral-small","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.ministral-3-14b-instruct":{"mode":"chat","base_model":"ministral-3-14b-instruct","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.ministral-3-3b-instruct":{"mode":"chat","base_model":"ministral-3-3b-instruct","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.ministral-3-8b-instruct":{"mode":"chat","base_model":"ministral-3-8b-instruct","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.mistral-7b-instruct-v0:2":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8191}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"mistral.mistral-large-2402-v1:0":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.mistral-large-2407-v1:0":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"supports_function_calling":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.mistral-large-3-675b-instruct":{"mode":"chat","base_model":"mistral-large-3-675b-instruct","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.mistral-small-2402-v1:0":{"mode":"chat","base_model":"mistral-small","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.mixtral-8x7b-instruct-v0:1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://aws.amazon.com/bedrock/pricing/","supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.voxtral-mini-3b-2507":{"mode":"chat","base_model":"voxtral-mini-3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":true,"supports_system_messages":true,"supports_native_structured_output":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral.voxtral-small-24b-2507":{"mode":"chat","base_model":"voxtral-small-24b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":true,"supports_system_messages":true,"supports_native_structured_output":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/codestral-2405":{"mode":"chat","base_model":"codestral","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/codestral-2508":{"mode":"chat","base_model":"codestral","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://mistral.ai/news/codestral-25-08","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/codestral-latest":{"mode":"chat","base_model":"codestral","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://docs.mistral.ai/models/model-cards/codestral-25-08","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8191,"range":{"min":1,"max":8191}}]},"mistral/codestral-mamba-latest":{"mode":"chat","base_model":"codestral-mamba","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://mistral.ai/technology/","supports_assistant_prefill":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/devstral-medium-2507":{"mode":"chat","base_model":"devstral-medium","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://mistral.ai/news/devstral","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/devstral-small-2505":{"mode":"chat","base_model":"devstral-small","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://mistral.ai/news/devstral","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/devstral-small-2507":{"mode":"chat","base_model":"devstral-small","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://mistral.ai/news/devstral","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/labs-devstral-small-2512":{"mode":"chat","base_model":"labs-devstral-small","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.mistral.ai/models/devstral-small-2-25-12","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/devstral-2512":{"mode":"chat","base_model":"devstral","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://mistral.ai/news/devstral-2-vibe-cli","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/magistral-medium-2506":{"mode":"chat","base_model":"magistral-medium","max_input_tokens":40000,"max_output_tokens":40000,"max_tokens":40000,"source":"https://mistral.ai/news/magistral","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/magistral-medium-2509":{"mode":"chat","base_model":"magistral-medium","max_input_tokens":40000,"max_output_tokens":40000,"max_tokens":40000,"source":"https://mistral.ai/news/magistral","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-ocr-latest":{"mode":"ocr","base_model":"mistral-ocr","supported_endpoints":["/v1/ocr"],"source":"https://mistral.ai/pricing#api-pricing","provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-ocr-2505-completion":{"mode":"ocr","base_model":"mistral-ocr-2505-completion","supported_endpoints":["/v1/ocr"],"source":"https://mistral.ai/pricing#api-pricing","provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/magistral-medium-latest":{"mode":"chat","base_model":"magistral-medium","max_input_tokens":40000,"max_output_tokens":40000,"max_tokens":40000,"source":"https://mistral.ai/news/magistral","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/magistral-small-2506":{"mode":"chat","base_model":"magistral-small","max_input_tokens":40000,"max_output_tokens":40000,"max_tokens":40000,"source":"https://mistral.ai/pricing#api-pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/magistral-small-latest":{"mode":"chat","base_model":"magistral-small","max_input_tokens":40000,"max_output_tokens":40000,"max_tokens":40000,"source":"https://mistral.ai/pricing#api-pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-embed":{"mode":"embedding","base_model":"mistral-embed","max_input_tokens":8192,"max_tokens":8192,"provider":"mistral"},"mistral/codestral-embed":{"mode":"embedding","base_model":"codestral-embed","max_input_tokens":8192,"max_tokens":8192,"provider":"mistral"},"mistral/codestral-embed-2505":{"mode":"embedding","base_model":"codestral-embed","max_input_tokens":8192,"max_tokens":8192,"provider":"mistral"},"mistral/mistral-large-2402":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-large-2407":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-large-2411":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-large-latest":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://docs.mistral.ai/models/mistral-large-3-25-12","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"mistral/mistral-large-3":{"mode":"chat","base_model":"mistral-large-3","max_input_tokens":256000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://docs.mistral.ai/models/mistral-large-3-25-12","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-medium":{"mode":"chat","base_model":"mistral-medium","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-medium-3-5-26-04","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-medium-2312":{"mode":"chat","base_model":"mistral-medium","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-medium-2505":{"mode":"chat","base_model":"mistral-medium","max_input_tokens":131072,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-medium-latest":{"mode":"chat","base_model":"mistral-medium","max_input_tokens":131072,"max_output_tokens":8191,"max_tokens":8191,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-medium-3-5-26-04","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-small":{"mode":"chat","base_model":"mistral-small","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/mistral-small-latest":{"mode":"chat","base_model":"mistral-small","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-small-4-0-26-03","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8191}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"mistral/mistral-tiny":{"mode":"chat","base_model":"mistral-tiny","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/open-codestral-mamba":{"mode":"chat","base_model":"codestral-mamba","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://mistral.ai/technology/","supports_assistant_prefill":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/open-mistral-7b":{"mode":"chat","base_model":"mistral-7b","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8191}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"mistral/open-mistral-nemo":{"mode":"chat","base_model":"mistral-nemo","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://mistral.ai/technology/","supports_assistant_prefill":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"mistral/open-mistral-nemo-2407":{"mode":"chat","base_model":"mistral-nemo","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://mistral.ai/technology/","supports_assistant_prefill":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/open-mixtral-8x22b":{"mode":"chat","base_model":"mixtral-8x22b","max_input_tokens":65336,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8191}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"mistral/open-mixtral-8x7b":{"mode":"chat","base_model":"mixtral-8x7b","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8191}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"mistral/pixtral-12b-2409":{"mode":"chat","base_model":"pixtral-12b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"mistral/pixtral-large-2411":{"mode":"chat","base_model":"pixtral-large","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"mistral/pixtral-large-latest":{"mode":"chat","base_model":"pixtral-large","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"moonshot.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-k2-0711-preview":{"mode":"chat","base_model":"kimi-k2","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.moonshot.ai/docs/pricing/chat#generation-model-kimi-k2","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-k2-0905-preview":{"mode":"chat","base_model":"kimi-k2","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.moonshot.ai/docs/pricing/chat#generation-model-kimi-k2","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-k2-turbo-preview":{"mode":"chat","base_model":"kimi-k2-turbo","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.moonshot.ai/docs/pricing/chat#generation-model-kimi-k2","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"moonshot","supported_endpoints":["/v1/chat/completions","/v1/batch"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_prompt_caching":true,"supports_web_search":true,"model":"kimi-k2.5","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-latest":{"mode":"chat","base_model":"kimi","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-latest-128k":{"mode":"chat","base_model":"kimi-latest-128k","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-latest-32k":{"mode":"chat","base_model":"kimi-latest-32k","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-latest-8k":{"mode":"chat","base_model":"kimi-latest-8k","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-thinking-preview":{"mode":"chat","base_model":"kimi-thinking","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.moonshot.ai/docs/pricing/chat#generation-model-kimi-k2","supports_vision":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.moonshot.ai/docs/pricing/chat#generation-model-kimi-k2","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/kimi-k2-thinking-turbo":{"mode":"chat","base_model":"kimi-k2-thinking-turbo","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.moonshot.ai/docs/pricing/chat#generation-model-kimi-k2","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-128k":{"mode":"chat","base_model":"moonshot-v1-128k","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-128k-0430":{"mode":"chat","base_model":"moonshot-v1-128k","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-128k-vision-preview":{"mode":"chat","base_model":"moonshot-v1-128k-vision","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-32k":{"mode":"chat","base_model":"moonshot-v1-32k","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-32k-0430":{"mode":"chat","base_model":"moonshot-v1-32k","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-32k-vision-preview":{"mode":"chat","base_model":"moonshot-v1-32k-vision","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-8k":{"mode":"chat","base_model":"moonshot-v1-8k","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-8k-0430":{"mode":"chat","base_model":"moonshot-v1-8k","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-8k-vision-preview":{"mode":"chat","base_model":"moonshot-v1-8k-vision","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"moonshot/moonshot-v1-auto":{"mode":"chat","base_model":"moonshot-v1-auto","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.moonshot.ai/docs/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"moonshot","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"morph/morph-v3-fast":{"mode":"chat","base_model":"morph-v3-fast","max_input_tokens":16000,"max_output_tokens":16000,"max_tokens":16000,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":false,"provider":"morph","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"morph/morph-v3-large":{"mode":"chat","base_model":"morph-v3-large","max_input_tokens":16000,"max_output_tokens":16000,"max_tokens":16000,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":false,"provider":"morph","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"multimodalembedding":{"mode":"embedding","base_model":"multimodalembedding","max_input_tokens":2048,"max_tokens":2048,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models","supported_endpoints":["/v1/embeddings"],"supported_modalities":["text","image","video"],"provider":"vertex_ai"},"multimodalembedding@001":{"mode":"embedding","base_model":"multimodalembedding","deprecation_date":"2027-04-01","max_input_tokens":2048,"max_tokens":2048,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models","supported_endpoints":["/v1/embeddings"],"supported_modalities":["text","image","video"],"provider":"vertex_ai"},"nscale/Qwen/QwQ-32B":{"mode":"chat","base_model":"qwq-32b","source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"nscale/Qwen/Qwen2.5-Coder-32B-Instruct":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct","source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/Qwen/Qwen2.5-Coder-3B-Instruct":{"mode":"chat","base_model":"qwen2.5-coder-3b-instruct","source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/Qwen/Qwen2.5-Coder-7B-Instruct":{"mode":"chat","base_model":"qwen2.5-coder-7b-instruct","source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"nscale/black-forest-labs/FLUX.1-schnell":{"mode":"image_generation","base_model":"flux.1-schnell","source":"https://docs.nscale.com/docs/inference/serverless-models/current#image-models","supported_endpoints":["/v1/images/generations"],"provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-70B":{"mode":"chat","base_model":"deepseek-r1-distill-llama-70b","metadata":{"notes":"Pricing listed as $0.75/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-8B":{"mode":"chat","base_model":"deepseek-r1-distill-llama-8b","metadata":{"notes":"Pricing listed as $0.05/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-1.5b","metadata":{"notes":"Pricing listed as $0.18/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-14b","metadata":{"notes":"Pricing listed as $0.14/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-32b","metadata":{"notes":"Pricing listed as $0.30/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-7B":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-7b","metadata":{"notes":"Pricing listed as $0.40/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/meta-llama/Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","metadata":{"notes":"Pricing listed as $0.06/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","metadata":{"notes":"Pricing listed as $0.40/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"nscale/mistralai/mixtral-8x22b-instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x22b-instruct","metadata":{"notes":"Pricing listed as $1.20/1M tokens total. Assumed 50/50 split for input/output."},"source":"https://docs.nscale.com/docs/inference/serverless-models/current#chat-models","provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"nscale/stabilityai/stable-diffusion-xl-base-1.0":{"mode":"image_generation","base_model":"stable-diffusion-xl-base-1.0","source":"https://docs.nscale.com/docs/inference/serverless-models/current#image-models","supported_endpoints":["/v1/images/generations"],"provider":"nscale","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nvidia.nemotron-nano-12b-v2":{"mode":"chat","base_model":"nvidia-nemotron-nano-12b-v2","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nvidia.nemotron-nano-9b-v2":{"mode":"chat","base_model":"nvidia-nemotron-nano-9b-v2","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nvidia.nemotron-nano-3-30b":{"mode":"chat","base_model":"nvidia-nemotron-nano-3-30b","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_native_structured_output":true,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"o1":{"mode":"chat","base_model":"o1","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["low","medium","high","xhigh"],"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":true},"o1-2024-12-17":{"mode":"chat","base_model":"o1","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["low","medium","high","xhigh"],"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":true},"o1-mini":{"mode":"chat","base_model":"o1-mini","max_input_tokens":128000,"max_output_tokens":65536,"max_tokens":65536,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"o1-mini-2024-09-12":{"mode":"chat","base_model":"o1-mini","deprecation_date":"2025-10-27","max_input_tokens":128000,"max_output_tokens":65536,"max_tokens":65536,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"o1-preview":{"mode":"chat","base_model":"o1","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}]},"o1-preview-2024-09-12":{"mode":"chat","base_model":"o1","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}]},"o1-pro":{"mode":"responses","base_model":"o1-pro","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":false,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["low","medium","high"],"supports_web_search":false,"supports_service_tier":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":false},"o1-pro-2025-03-19":{"mode":"responses","base_model":"o1-pro","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":false,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["low","medium","high"],"supports_web_search":false,"supports_service_tier":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":false},"o3":{"mode":"chat","base_model":"o3","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses","/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["low","medium","high","xhigh"],"supports_assistant_prefill":true,"supports_system_messages":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":true},"o3-2025-04-16":{"mode":"chat","base_model":"o3","deprecation_date":"2026-12-11","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses","/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["low","medium","high","xhigh"],"supports_assistant_prefill":true,"supports_system_messages":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":true},"o3-deep-research":{"mode":"responses","base_model":"o3","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"o3-deep-research-2025-06-26":{"mode":"responses","base_model":"o3","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"o3-mini":{"mode":"chat","base_model":"o3-mini","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["low","medium","high","xhigh"],"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true,"supports_system_messages":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":true},"o3-mini-2025-01-31":{"mode":"chat","base_model":"o3-mini","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","minElements":1,"maxElements":4}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["low","medium","high","xhigh"],"supports_web_search":false,"supports_service_tier":true,"supports_assistant_prefill":true,"supports_system_messages":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":true},"o3-pro":{"mode":"responses","base_model":"o3-pro","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["low","medium","high"],"supports_service_tier":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":false},"o3-pro-2025-06-10":{"mode":"responses","base_model":"o3-pro","deprecation_date":"2026-12-11","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"reasoning_effort_levels":["low","medium","high"],"supports_service_tier":false,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":false},"o4-mini":{"mode":"chat","base_model":"o4-mini","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["low","medium","high","xhigh"],"supports_assistant_prefill":true,"supports_system_messages":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":true},"o4-mini-2025-04-16":{"mode":"chat","base_model":"o4-mini","deprecation_date":"2026-10-23","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["low","medium","high","xhigh"],"supports_assistant_prefill":true,"supports_system_messages":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":true},"o4-mini-deep-research":{"mode":"responses","base_model":"o4-mini","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"o4-mini-deep-research-2025-06-26":{"mode":"responses","base_model":"o4-mini","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}]},"oci/meta.llama-3.1-405b-instruct":{"mode":"chat","base_model":"llama-3.1-405b-instruct","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/meta.llama-3.2-90b-vision-instruct":{"mode":"chat","base_model":"llama-3.2-90b-vision-instruct","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"supports_vision":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/meta.llama-3.3-70b-instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/meta.llama-4-maverick-17b-128e-instruct-fp8":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct-fp8","max_input_tokens":512000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"supports_vision":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/meta.llama-4-scout-17b-16e-instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_input_tokens":192000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"oci/xai.grok-3":{"mode":"chat","base_model":"grok-3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/xai.grok-3-fast":{"mode":"chat","base_model":"grok-3-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/xai.grok-3-mini":{"mode":"chat","base_model":"grok-3-mini","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/xai.grok-3-mini-fast":{"mode":"chat","base_model":"grok-3-mini-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/xai.grok-4":{"mode":"chat","base_model":"grok-4","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/cohere.command-latest":{"mode":"chat","base_model":"command","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/cloud/ai/generative-ai/pricing/","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/cohere.command-a-03-2025":{"mode":"chat","base_model":"command-a","max_input_tokens":256000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/cloud/ai/generative-ai/pricing/","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"oci/cohere.command-plus-latest":{"mode":"chat","base_model":"command-plus","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/cloud/ai/generative-ai/pricing/","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/codegeex4":{"mode":"chat","base_model":"codegeex4","max_input_tokens":32768,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":false,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/codegemma":{"mode":"completion","base_model":"codegemma","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/codellama":{"mode":"completion","base_model":"codellama","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/deepseek-coder-v2-base":{"mode":"completion","base_model":"deepseek-coder-v2-base","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/deepseek-coder-v2-instruct":{"mode":"chat","base_model":"deepseek-coder-v2-instruct","max_input_tokens":32768,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/deepseek-coder-v2-lite-base":{"mode":"completion","base_model":"deepseek-coder-v2-lite-base","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/deepseek-coder-v2-lite-instruct":{"mode":"chat","base_model":"deepseek-coder-v2-lite-instruct","max_input_tokens":32768,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/deepseek-v3.1:671b-cloud":{"mode":"chat","base_model":"deepseek-v3.1-671b","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/gpt-oss:120b-cloud":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/gpt-oss:20b-cloud":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/internlm2_5-20b-chat":{"mode":"chat","base_model":"internlm2-5-20b-chat","max_input_tokens":32768,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/llama2":{"mode":"chat","base_model":"llama-2","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/llama2-uncensored":{"mode":"completion","base_model":"llama-2-uncensored","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/llama2:13b":{"mode":"chat","base_model":"llama-2-13b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/llama2:70b":{"mode":"chat","base_model":"llama-2-70b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/llama2:7b":{"mode":"chat","base_model":"llama-2-7b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/llama3":{"mode":"chat","base_model":"llama-3","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/llama3.1":{"mode":"chat","base_model":"llama-3.1","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/llama3:70b":{"mode":"chat","base_model":"llama-3-70b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/llama3:8b":{"mode":"chat","base_model":"llama-3-8b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/mistral":{"mode":"completion","base_model":"mistral","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/mistral-7B-Instruct-v0.1":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/mistral-7B-Instruct-v0.2":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/mistral-large-instruct-2407":{"mode":"chat","base_model":"mistral-large-instruct","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/mixtral-8x22B-Instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x22b-instruct","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":65536}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/mixtral-8x7B-Instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/orca-mini":{"mode":"completion","base_model":"orca-mini","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ollama/qwen3-coder:480b-cloud":{"mode":"chat","base_model":"qwen3-coder-480b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"provider":"ollama","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ollama/vicuna":{"mode":"completion","base_model":"vicuna","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"provider":"ollama","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"omni-moderation-2024-09-26":{"mode":"moderation","base_model":"omni-moderation","max_input_tokens":32768,"max_output_tokens":0,"max_tokens":0,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"omni-moderation-latest":{"mode":"moderation","base_model":"omni-moderation","max_input_tokens":32768,"max_output_tokens":0,"max_tokens":0,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"omni-moderation-latest-intents":{"mode":"moderation","base_model":"omni-moderation-latest-intents","max_input_tokens":32768,"max_output_tokens":0,"max_tokens":0,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai.gpt-oss-120b-1:0":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"openai.gpt-oss-20b-1:0":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"openai.gpt-oss-safeguard-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"openai.gpt-oss-safeguard-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"openrouter/anthropic/claude-3-haiku":{"mode":"chat","base_model":"claude-3-haiku","max_tokens":200000,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"max_input_tokens":200000,"max_output_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-3.5-sonnet":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openrouter","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-3.7-sonnet":{"mode":"chat","base_model":"claude-3-7-sonnet","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openrouter","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-opus-4":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-opus-4.1":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_response_schema":false,"supports_web_search":true,"provider":"openrouter","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-sonnet-4":{"mode":"chat","base_model":"claude-sonnet-4","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_response_schema":false,"supports_web_search":true,"provider":"openrouter","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-opus-4.5":{"mode":"chat","base_model":"claude-opus-4-5","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_response_schema":true,"supports_web_search":true,"provider":"openrouter","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-sonnet-4.5":{"mode":"chat","base_model":"claude-sonnet-4-5","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_response_schema":true,"supports_web_search":true,"provider":"openrouter","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-haiku-4.5":{"mode":"chat","base_model":"claude-haiku-4-5","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_response_schema":true,"supports_web_search":true,"provider":"openrouter","tool_use_system_prompt_tokens":346,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/bytedance/ui-tars-1.5-7b":{"mode":"chat","base_model":"ui-tars-1.5-7b","max_input_tokens":131072,"max_output_tokens":2048,"max_tokens":2048,"source":"https://openrouter.ai/api/v1/models/bytedance/ui-tars-1.5-7b","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-chat":{"mode":"chat","base_model":"deepseek-chat","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"supports_prompt_caching":true,"supports_tool_choice":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-chat-v3-0324":{"mode":"chat","base_model":"deepseek-chat-v3","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-chat-v3.1":{"mode":"chat","base_model":"deepseek-chat","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-v3.2":{"mode":"chat","base_model":"deepseek","deprecation_date":"2026-09-28","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-v3.2-exp":{"mode":"chat","base_model":"deepseek-v3.2","deprecation_date":"2026-09-28","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"source":"https://openrouter.ai/api/v1/models","supports_assistant_prefill":true,"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-r1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":65336,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_assistant_prefill":true,"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-r1-0528":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":65336,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_assistant_prefill":true,"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemini-2.0-flash-001":{"mode":"chat","base_model":"gemini-2.0-flash","deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":30,"max_tokens":8192,"max_video_length":1,"max_videos_per_prompt":10,"supports_audio_output":true,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"openrouter/google/gemini-2.5-flash":{"mode":"chat","base_model":"gemini-2.5-flash","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"supports_audio_output":true,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_image_size":false,"deprecation_date":"2026-10-20","supports_prompt_caching":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_pdf_input":true,"supports_reasoning":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemini-2.5-pro":{"mode":"chat","base_model":"gemini-2.5-pro","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"supports_audio_output":true,"supports_function_calling":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"deprecation_date":"2026-10-20","supports_prompt_caching":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_pdf_input":true,"supports_reasoning":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemini-3-pro-preview":{"mode":"chat","base_model":"gemini-3-pro","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemini-3-flash-preview":{"mode":"chat","base_model":"gemini-3-flash","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":2000,"source":"https://ai.google.dev/pricing/gemini-3","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":800000,"supports_video_input":true,"provider":"openrouter","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/gryphe/mythomax-l2-13b":{"mode":"chat","base_model":"mythomax-l2-13b","max_input_tokens":4096,"max_output_tokens":3686,"max_tokens":8192,"supports_tool_choice":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mancer/weaver":{"mode":"chat","base_model":"weaver","max_tokens":8000,"supports_tool_choice":true,"max_input_tokens":8000,"max_output_tokens":6000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/meta-llama/llama-3-70b-instruct":{"mode":"chat","base_model":"llama-3-70b-instruct","max_tokens":8192,"supports_tool_choice":true,"max_input_tokens":8192,"max_output_tokens":8000,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/minimax/minimax-m2":{"mode":"chat","base_model":"minimax-m2","max_input_tokens":204800,"max_output_tokens":204800,"max_tokens":204800,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":204800}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/devstral-2512":{"mode":"chat","base_model":"devstral","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/ministral-3b-2512":{"mode":"chat","base_model":"ministral-3b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/ministral-8b-2512":{"mode":"chat","base_model":"ministral-8b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/ministral-14b-2512":{"mode":"chat","base_model":"ministral-14b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mistral-large-2512":{"mode":"chat","base_model":"mistral-large","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_vision":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mistral-7b-instruct":{"mode":"chat","base_model":"mistral-7b-instruct","max_tokens":8192,"supports_tool_choice":true,"max_input_tokens":32768,"max_output_tokens":8191,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mistral-large":{"mode":"chat","base_model":"mistral-large","max_tokens":32000,"supports_tool_choice":true,"max_input_tokens":128000,"max_output_tokens":102400,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mistral-small-3.1-24b-instruct":{"mode":"chat","base_model":"mistral-small-3.1-24b-instruct","max_tokens":32000,"supports_tool_choice":true,"max_input_tokens":128000,"max_output_tokens":102400,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mistral-small-3.2-24b-instruct":{"mode":"chat","base_model":"mistral-small-3.2-24b-instruct","max_tokens":32000,"supports_tool_choice":true,"max_input_tokens":256000,"max_output_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mixtral-8x22b-instruct":{"mode":"chat","base_model":"mixtral-8x22b-instruct","max_tokens":65536,"supports_tool_choice":true,"max_input_tokens":65536,"max_output_tokens":52428,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/moonshotai/kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/moonshotai/kimi-k2.5","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-3.5-turbo":{"mode":"chat","base_model":"gpt-3.5-turbo","max_tokens":4095,"supports_tool_choice":true,"max_input_tokens":16385,"max_output_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":4095}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-3.5-turbo-16k":{"mode":"chat","base_model":"gpt-3.5-turbo-16k","max_tokens":16383,"supports_tool_choice":true,"max_input_tokens":16385,"max_output_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16383}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4":{"mode":"chat","base_model":"gpt-4","max_tokens":8192,"supports_tool_choice":true,"max_input_tokens":8191,"max_output_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4.1-mini":{"mode":"chat","base_model":"gpt-4.1-mini","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4.1-nano":{"mode":"chat","base_model":"gpt-4.1-nano","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4o":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_prompt_caching":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_reasoning":false,"supports_response_schema":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4o-2024-05-13":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5-chat":{"mode":"chat","base_model":"gpt-5-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_reasoning":true,"supports_tool_choice":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5-codex":{"mode":"chat","base_model":"gpt-5-codex","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_reasoning":true,"supports_tool_choice":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5.2-codex":{"mode":"chat","base_model":"gpt-5.2-codex","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5":{"mode":"chat","base_model":"gpt-5","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5.2":{"mode":"chat","base_model":"gpt-5.2","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5.2-chat":{"mode":"chat","base_model":"gpt-5.2-chat","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-5.2-pro":{"mode":"chat","base_model":"gpt-5.2-pro","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/openai/gpt-oss-120b","supports_audio_input":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/openai/gpt-oss-20b","supports_audio_input":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/o1":{"mode":"chat","base_model":"o1","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":100000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/o3-mini":{"mode":"chat","base_model":"o3-mini","max_input_tokens":128000,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"supports_prompt_caching":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_response_schema":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":65536}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/o3-mini-high":{"mode":"chat","base_model":"o3-mini-high","max_input_tokens":128000,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"supports_prompt_caching":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_response_schema":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":65536}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen-2.5-coder-32b-instruct":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct","max_input_tokens":33792,"max_output_tokens":33792,"max_tokens":33792,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":33792}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen-vl-plus":{"mode":"chat","base_model":"qwen-vl-plus","max_input_tokens":8192,"max_output_tokens":2048,"max_tokens":2048,"supports_tool_choice":true,"supports_vision":true,"provider":"openrouter","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-coder":{"mode":"chat","base_model":"qwen3-coder","max_input_tokens":262100,"max_output_tokens":262100,"max_tokens":262100,"source":"https://openrouter.ai/qwen/qwen3-coder","supports_audio_input":false,"supports_tool_choice":true,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":262100}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-235b-a22b-2507":{"mode":"chat","base_model":"qwen3-235b-a22b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/qwen/qwen3-235b-a22b-2507","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-235b-a22b-thinking-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/qwen/qwen3-235b-a22b-thinking-2507","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/switchpoint/router":{"mode":"chat","base_model":"router","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/switchpoint/router","supports_tool_choice":true,"provider":"openrouter","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/undi95/remm-slerp-l2-13b":{"mode":"chat","base_model":"remm-slerp-l2-13b","max_tokens":6144,"supports_tool_choice":true,"max_input_tokens":6144,"max_output_tokens":5529,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":6144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/x-ai/grok-4":{"mode":"chat","base_model":"grok-4","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://openrouter.ai/x-ai/grok-4","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/z-ai/glm-4.6":{"mode":"chat","base_model":"glm-4.6","max_input_tokens":202800,"max_output_tokens":131000,"max_tokens":131000,"source":"https://openrouter.ai/z-ai/glm-4.6","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_pdf_input":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/z-ai/glm-4.6:exacto":{"mode":"chat","base_model":"glm-4.6","max_input_tokens":202800,"max_output_tokens":131000,"max_tokens":131000,"source":"https://openrouter.ai/z-ai/glm-4.6:exacto","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/xiaomi/mimo-v2-flash":{"mode":"chat","base_model":"mimo-v2-flash","max_input_tokens":262144,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"supports_prompt_caching":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/z-ai/glm-4.7":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":202752,"max_output_tokens":64000,"max_tokens":64000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":true,"supports_prompt_caching":false,"supports_assistant_prefill":true,"supports_audio_input":false,"supports_pdf_input":false,"supports_response_schema":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/z-ai/glm-4.7-flash":{"mode":"chat","base_model":"glm-4.7-flash","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":true,"supports_prompt_caching":false,"supports_response_schema":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/minimax/minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","deprecation_date":"2026-10-08","max_input_tokens":204000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":true,"supports_prompt_caching":false,"supports_computer_use":false,"supports_pdf_input":false,"supports_response_schema":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/DeepSeek-R1-Distill-Llama-70B":{"mode":"chat","base_model":"deepseek-r1-distill-llama-70b","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"source":"https://endpoints.ai.cloud.ovh.net/models/deepseek-r1-distill-llama-70b","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"ovhcloud","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ovhcloud/Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"source":"https://endpoints.ai.cloud.ovh.net/models/llama-3-1-8b-instruct","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/Meta-Llama-3_1-70B-Instruct":{"mode":"chat","base_model":"llama-3-1-70b-instruct","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"source":"https://endpoints.ai.cloud.ovh.net/models/meta-llama-3-1-70b-instruct","supports_function_calling":false,"supports_response_schema":false,"supports_tool_choice":false,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/Meta-Llama-3_3-70B-Instruct":{"mode":"chat","base_model":"llama-3-3-70b-instruct","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"source":"https://endpoints.ai.cloud.ovh.net/models/meta-llama-3-3-70b-instruct","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/Mistral-7B-Instruct-v0.3":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":127000,"max_output_tokens":127000,"max_tokens":127000,"source":"https://endpoints.ai.cloud.ovh.net/models/mistral-7b-instruct-v0-3","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":127000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ovhcloud/Mistral-Nemo-Instruct-2407":{"mode":"chat","base_model":"mistral-nemo-instruct","max_input_tokens":118000,"max_output_tokens":118000,"max_tokens":118000,"source":"https://endpoints.ai.cloud.ovh.net/models/mistral-nemo-instruct-2407","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":118000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/Mistral-Small-3.2-24B-Instruct-2506":{"mode":"chat","base_model":"mistral-small-3.2-24b-instruct","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://endpoints.ai.cloud.ovh.net/models/mistral-small-3-2-24b-instruct-2506","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/Mixtral-8x7B-Instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://endpoints.ai.cloud.ovh.net/models/mixtral-8x7b-instruct-v0-1","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/Qwen2.5-Coder-32B-Instruct":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://endpoints.ai.cloud.ovh.net/models/qwen2-5-coder-32b-instruct","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/Qwen2.5-VL-72B-Instruct":{"mode":"chat","base_model":"qwen2.5-vl-72b-instruct","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://endpoints.ai.cloud.ovh.net/models/qwen2-5-vl-72b-instruct","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/Qwen3-32B":{"mode":"chat","base_model":"qwen3-32b","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://endpoints.ai.cloud.ovh.net/models/qwen3-32b","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ovhcloud/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"source":"https://endpoints.ai.cloud.ovh.net/models/gpt-oss-120b","supports_function_calling":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ovhcloud/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"source":"https://endpoints.ai.cloud.ovh.net/models/gpt-oss-20b","supports_function_calling":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"ovhcloud/llava-v1.6-mistral-7b-hf":{"mode":"chat","base_model":"llava-v1.6-mistral-7b","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://endpoints.ai.cloud.ovh.net/models/llava-next-mistral-7b","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"ovhcloud/mamba-codestral-7B-v0.1":{"mode":"chat","base_model":"mamba-codestral-7b","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://endpoints.ai.cloud.ovh.net/models/mamba-codestral-7b-v0-1","supports_function_calling":false,"supports_response_schema":true,"supports_tool_choice":false,"provider":"ovhcloud","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"palm/chat-bison":{"mode":"chat","base_model":"chat-bison","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"palm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"palm/chat-bison-001":{"mode":"chat","base_model":"chat-bison-001","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"palm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"palm/text-bison":{"mode":"completion","base_model":"text-bison","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"palm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"palm/text-bison-001":{"mode":"completion","base_model":"text-bison-001","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"palm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"palm/text-bison-safety-off":{"mode":"completion","base_model":"text-bison-safety-off","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"palm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"palm/text-bison-safety-recitation-off":{"mode":"completion","base_model":"text-bison-safety-recitation-off","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"palm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"parallel_ai/search":{"mode":"search","base_model":"search","provider":"parallel_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"parallel_ai/search-pro":{"mode":"search","base_model":"search-pro","provider":"parallel_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/codellama-34b-instruct":{"mode":"chat","base_model":"codellama-34b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"perplexity/codellama-70b-instruct":{"mode":"chat","base_model":"codellama-70b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"perplexity/llama-2-70b-chat":{"mode":"chat","base_model":"llama-2-70b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"perplexity/llama-3.1-70b-instruct":{"mode":"chat","base_model":"llama-3.1-70b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/llama-3.1-8b-instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/llama-3.1-sonar-huge-128k-online":{"mode":"chat","base_model":"llama-3.1-sonar-huge-128k-online","deprecation_date":"2025-02-22","max_input_tokens":127072,"max_output_tokens":127072,"max_tokens":127072,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":127072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"perplexity/llama-3.1-sonar-large-128k-chat":{"mode":"chat","base_model":"llama-3.1-sonar-large-128k-chat","deprecation_date":"2025-02-22","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"perplexity/llama-3.1-sonar-large-128k-online":{"mode":"chat","base_model":"llama-3.1-sonar-large-128k-online","deprecation_date":"2025-02-22","max_input_tokens":127072,"max_output_tokens":127072,"max_tokens":127072,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":127072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"perplexity/llama-3.1-sonar-small-128k-chat":{"mode":"chat","base_model":"llama-3.1-sonar-small-128k-chat","deprecation_date":"2025-02-22","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"perplexity/llama-3.1-sonar-small-128k-online":{"mode":"chat","base_model":"llama-3.1-sonar-small-128k-online","deprecation_date":"2025-02-22","max_input_tokens":127072,"max_output_tokens":127072,"max_tokens":127072,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":127072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"perplexity/mistral-7b-instruct":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"perplexity/mixtral-8x7b-instruct":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/pplx-70b-chat":{"mode":"chat","base_model":"pplx-70b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/pplx-70b-online":{"mode":"chat","base_model":"pplx-70b-online","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/pplx-7b-chat":{"mode":"chat","base_model":"pplx-7b-chat","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/pplx-7b-online":{"mode":"chat","base_model":"pplx-7b-online","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/sonar":{"mode":"chat","base_model":"sonar","max_input_tokens":128000,"max_tokens":128000,"supports_web_search":true,"provider":"perplexity","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.2,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.9,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"search_mode","label":"Search Mode","helpText":"Controls search mode: 'academic' prioritizes scholarly sources, 'sec' prioritizes SEC filings, 'web' uses general web search.","type":"select","default":"web","options":[{"label":"Academic","value":"academic"},{"label":"SEC","value":"sec"},{"label":"Web","value":"web"}]},{"id":"search_domain_filter","label":"Search Domain Filter","helpText":"A comma-separated list of domains to limit search results to. Add a '-' at the beginning of a domain to exclude it. Limited to 20 domains.","type":"text"},{"id":"return_images","label":"Return Images","helpText":"Determines whether search results should include images.","type":"boolean","default":false},{"id":"return_videos","label":"Return Videos","helpText":"Determines whether search results should include videos.","type":"boolean","default":false},{"id":"return_related_questions","label":"Return Related Questions","helpText":"Determines whether related questions should be returned.","type":"boolean","default":false},{"id":"search_recency_filter","label":"Search Recency Filter","helpText":"Filters search results based on time (e.g., 'week', 'day').","type":"text"},{"id":"search_after_date_filter","label":"Search After Date Filter","helpText":"Filters search results to only include content published after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"search_before_date_filter","label":"Search Before Date Filter","helpText":"Filters search results to only include content published before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_after_filter","label":"Last Updated After Filter","helpText":"Filters search results to only include content last updated after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_before_filter","label":"Last Updated Before Filter","helpText":"Filters search results to only include content last updated before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"top_k","label":"Top K","helpText":"The number of tokens to keep for top-k filtering. Limits the model to consider only the k most likely next tokens at each step. A value of 0 disables this filter.","type":"number","default":0},{"id":"disable_search","label":"Disable Search","helpText":"When set to true, disables web search completely and the model will only use its training data to respond.","type":"boolean","default":false},{"id":"enable_search_classifier","label":"Enable Search Classifier","helpText":"Enables a classifier that decides if web search is needed based on your query.","type":"boolean","default":false},{"id":"search_context_size","label":"Search Context Size","helpText":"Controls the size of search context used in web search. Options: low, medium, high.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"language_preference","label":"Language Preference","helpText":"Specifies the preferred language for the chat completion response (e.g., English, Korean, Spanish, etc.).","type":"text"}]},"perplexity/sonar-deep-research":{"mode":"chat","base_model":"sonar","max_input_tokens":128000,"max_tokens":128000,"supports_reasoning":true,"supports_web_search":true,"provider":"perplexity","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.2,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.9,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"search_mode","label":"Search Mode","helpText":"Controls search mode: 'academic' prioritizes scholarly sources, 'sec' prioritizes SEC filings, 'web' uses general web search.","type":"select","default":"web","options":[{"label":"Academic","value":"academic"},{"label":"SEC","value":"sec"},{"label":"Web","value":"web"}]},{"id":"search_domain_filter","label":"Search Domain Filter","helpText":"A comma-separated list of domains to limit search results to. Add a '-' at the beginning of a domain to exclude it. Limited to 20 domains.","type":"text"},{"id":"return_images","label":"Return Images","helpText":"Determines whether search results should include images.","type":"boolean","default":false},{"id":"return_videos","label":"Return Videos","helpText":"Determines whether search results should include videos.","type":"boolean","default":false},{"id":"return_related_questions","label":"Return Related Questions","helpText":"Determines whether related questions should be returned.","type":"boolean","default":false},{"id":"search_recency_filter","label":"Search Recency Filter","helpText":"Filters search results based on time (e.g., 'week', 'day').","type":"text"},{"id":"search_after_date_filter","label":"Search After Date Filter","helpText":"Filters search results to only include content published after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"search_before_date_filter","label":"Search Before Date Filter","helpText":"Filters search results to only include content published before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_after_filter","label":"Last Updated After Filter","helpText":"Filters search results to only include content last updated after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_before_filter","label":"Last Updated Before Filter","helpText":"Filters search results to only include content last updated before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"top_k","label":"Top K","helpText":"The number of tokens to keep for top-k filtering. Limits the model to consider only the k most likely next tokens at each step. A value of 0 disables this filter.","type":"number","default":0},{"id":"disable_search","label":"Disable Search","helpText":"When set to true, disables web search completely and the model will only use its training data to respond.","type":"boolean","default":false},{"id":"enable_search_classifier","label":"Enable Search Classifier","helpText":"Enables a classifier that decides if web search is needed based on your query.","type":"boolean","default":false},{"id":"search_context_size","label":"Search Context Size","helpText":"Controls the size of search context used in web search. Options: low, medium, high.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Controls how much computational effort the AI dedicates to each query for deep research models. 'low' provides faster, simpler answers with reduced token usage, 'medium' offers a balanced approach, and 'high' delivers deeper, more thorough responses with increased token usage.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]}]},"perplexity/sonar-medium-chat":{"mode":"chat","base_model":"sonar-medium-chat","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/sonar-medium-online":{"mode":"chat","base_model":"sonar-medium-online","max_input_tokens":12000,"max_output_tokens":12000,"max_tokens":12000,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":12000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/sonar-pro":{"mode":"chat","base_model":"sonar-pro","max_input_tokens":200000,"max_output_tokens":8000,"max_tokens":8000,"supports_web_search":true,"provider":"perplexity","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.2,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.9,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"search_mode","label":"Search Mode","helpText":"Controls search mode: 'academic' prioritizes scholarly sources, 'sec' prioritizes SEC filings, 'web' uses general web search.","type":"select","default":"web","options":[{"label":"Academic","value":"academic"},{"label":"SEC","value":"sec"},{"label":"Web","value":"web"}]},{"id":"search_domain_filter","label":"Search Domain Filter","helpText":"A comma-separated list of domains to limit search results to. Add a '-' at the beginning of a domain to exclude it. Limited to 20 domains.","type":"text"},{"id":"return_images","label":"Return Images","helpText":"Determines whether search results should include images.","type":"boolean","default":false},{"id":"return_videos","label":"Return Videos","helpText":"Determines whether search results should include videos.","type":"boolean","default":false},{"id":"return_related_questions","label":"Return Related Questions","helpText":"Determines whether related questions should be returned.","type":"boolean","default":false},{"id":"search_recency_filter","label":"Search Recency Filter","helpText":"Filters search results based on time (e.g., 'week', 'day').","type":"text"},{"id":"search_after_date_filter","label":"Search After Date Filter","helpText":"Filters search results to only include content published after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"search_before_date_filter","label":"Search Before Date Filter","helpText":"Filters search results to only include content published before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_after_filter","label":"Last Updated After Filter","helpText":"Filters search results to only include content last updated after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_before_filter","label":"Last Updated Before Filter","helpText":"Filters search results to only include content last updated before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"top_k","label":"Top K","helpText":"The number of tokens to keep for top-k filtering. Limits the model to consider only the k most likely next tokens at each step. A value of 0 disables this filter.","type":"number","default":0},{"id":"disable_search","label":"Disable Search","helpText":"When set to true, disables web search completely and the model will only use its training data to respond.","type":"boolean","default":false},{"id":"enable_search_classifier","label":"Enable Search Classifier","helpText":"Enables a classifier that decides if web search is needed based on your query.","type":"boolean","default":false},{"id":"search_context_size","label":"Search Context Size","helpText":"Controls the size of search context used in web search. Options: low, medium, high.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"language_preference","label":"Language Preference","helpText":"Specifies the preferred language for the chat completion response (e.g., English, Korean, Spanish, etc.).","type":"text"}]},"perplexity/sonar-reasoning":{"mode":"chat","base_model":"sonar","max_input_tokens":128000,"max_tokens":128000,"supports_reasoning":true,"supports_web_search":true,"provider":"perplexity","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.2,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.9,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"search_mode","label":"Search Mode","helpText":"Controls search mode: 'academic' prioritizes scholarly sources, 'sec' prioritizes SEC filings, 'web' uses general web search.","type":"select","default":"web","options":[{"label":"Academic","value":"academic"},{"label":"SEC","value":"sec"},{"label":"Web","value":"web"}]},{"id":"search_domain_filter","label":"Search Domain Filter","helpText":"A comma-separated list of domains to limit search results to. Add a '-' at the beginning of a domain to exclude it. Limited to 20 domains.","type":"text"},{"id":"return_images","label":"Return Images","helpText":"Determines whether search results should include images.","type":"boolean","default":false},{"id":"return_videos","label":"Return Videos","helpText":"Determines whether search results should include videos.","type":"boolean","default":false},{"id":"return_related_questions","label":"Return Related Questions","helpText":"Determines whether related questions should be returned.","type":"boolean","default":false},{"id":"search_recency_filter","label":"Search Recency Filter","helpText":"Filters search results based on time (e.g., 'week', 'day').","type":"text"},{"id":"search_after_date_filter","label":"Search After Date Filter","helpText":"Filters search results to only include content published after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"search_before_date_filter","label":"Search Before Date Filter","helpText":"Filters search results to only include content published before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_after_filter","label":"Last Updated After Filter","helpText":"Filters search results to only include content last updated after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_before_filter","label":"Last Updated Before Filter","helpText":"Filters search results to only include content last updated before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"top_k","label":"Top K","helpText":"The number of tokens to keep for top-k filtering. Limits the model to consider only the k most likely next tokens at each step. A value of 0 disables this filter.","type":"number","default":0},{"id":"disable_search","label":"Disable Search","helpText":"When set to true, disables web search completely and the model will only use its training data to respond.","type":"boolean","default":false},{"id":"enable_search_classifier","label":"Enable Search Classifier","helpText":"Enables a classifier that decides if web search is needed based on your query.","type":"boolean","default":false},{"id":"search_context_size","label":"Search Context Size","helpText":"Controls the size of search context used in web search. Options: low, medium, high.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]}]},"perplexity/sonar-reasoning-pro":{"mode":"chat","base_model":"sonar-reasoning-pro","max_input_tokens":128000,"max_tokens":128000,"supports_reasoning":true,"supports_web_search":true,"provider":"perplexity","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.2,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.9,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"search_mode","label":"Search Mode","helpText":"Controls search mode: 'academic' prioritizes scholarly sources, 'sec' prioritizes SEC filings, 'web' uses general web search.","type":"select","default":"web","options":[{"label":"Academic","value":"academic"},{"label":"SEC","value":"sec"},{"label":"Web","value":"web"}]},{"id":"search_domain_filter","label":"Search Domain Filter","helpText":"A comma-separated list of domains to limit search results to. Add a '-' at the beginning of a domain to exclude it. Limited to 20 domains.","type":"text"},{"id":"return_images","label":"Return Images","helpText":"Determines whether search results should include images.","type":"boolean","default":false},{"id":"return_videos","label":"Return Videos","helpText":"Determines whether search results should include videos.","type":"boolean","default":false},{"id":"return_related_questions","label":"Return Related Questions","helpText":"Determines whether related questions should be returned.","type":"boolean","default":false},{"id":"search_recency_filter","label":"Search Recency Filter","helpText":"Filters search results based on time (e.g., 'week', 'day').","type":"text"},{"id":"search_after_date_filter","label":"Search After Date Filter","helpText":"Filters search results to only include content published after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"search_before_date_filter","label":"Search Before Date Filter","helpText":"Filters search results to only include content published before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_after_filter","label":"Last Updated After Filter","helpText":"Filters search results to only include content last updated after this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"last_updated_before_filter","label":"Last Updated Before Filter","helpText":"Filters search results to only include content last updated before this date. Format: MM/DD/YYYY (e.g., 3/1/2025).","type":"text"},{"id":"top_k","label":"Top K","helpText":"The number of tokens to keep for top-k filtering. Limits the model to consider only the k most likely next tokens at each step. A value of 0 disables this filter.","type":"number","default":0},{"id":"disable_search","label":"Disable Search","helpText":"When set to true, disables web search completely and the model will only use its training data to respond.","type":"boolean","default":false},{"id":"enable_search_classifier","label":"Enable Search Classifier","helpText":"Enables a classifier that decides if web search is needed based on your query.","type":"boolean","default":false},{"id":"search_context_size","label":"Search Context Size","helpText":"Controls the size of search context used in web search. Options: low, medium, high.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]}]},"perplexity/sonar-small-chat":{"mode":"chat","base_model":"sonar-small-chat","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/sonar-small-online":{"mode":"chat","base_model":"sonar-small-online","max_input_tokens":12000,"max_output_tokens":12000,"max_tokens":12000,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":12000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/swiss-ai/apertus-8b-instruct":{"mode":"chat","base_model":"apertus-8b-instruct","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/swiss-ai/apertus-70b-instruct":{"mode":"chat","base_model":"apertus-70b-instruct","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/aisingapore/Gemma-SEA-LION-v4-27B-IT":{"mode":"chat","base_model":"gemma-sea-lion-v4-27b-it","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/BSC-LT/salamandra-7b-instruct-tools-16k":{"mode":"chat","base_model":"salamandra-7b-instruct-tools-16k","max_input_tokens":16384,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/BSC-LT/ALIA-40b-instruct_Q8_0":{"mode":"chat","base_model":"alia-40b-instruct-q8-0","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/allenai/Olmo-3-7B-Instruct":{"mode":"chat","base_model":"olmo-3-7b-instruct","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/preset/pro-search":{"mode":"responses","base_model":"preset/pro-search","supports_web_search":true,"supports_function_calling":true,"provider":"perplexity","supports_preset":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/openai/gpt-4o":{"mode":"responses","base_model":"gpt-4o","supports_web_search":true,"supports_reasoning":false,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/openai/gpt-4o-mini":{"mode":"responses","base_model":"gpt-4o-mini","supports_web_search":true,"supports_reasoning":false,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/openai/gpt-5.2":{"mode":"responses","base_model":"gpt-5.2","supports_web_search":true,"supports_reasoning":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/anthropic/claude-3-5-sonnet-20241022":{"mode":"responses","base_model":"claude-3-5-sonnet","supports_web_search":true,"supports_reasoning":false,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/anthropic/claude-3-5-haiku-20241022":{"mode":"responses","base_model":"claude-3-5-haiku","supports_web_search":true,"supports_reasoning":false,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/google/gemini-2.0-flash-exp":{"mode":"responses","base_model":"gemini-2.0-flash","supports_web_search":true,"supports_reasoning":false,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/google/gemini-2.0-flash-thinking-exp":{"mode":"responses","base_model":"gemini-2.0-flash-thinking","supports_web_search":true,"supports_reasoning":true,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/xai/grok-2-1212":{"mode":"responses","base_model":"grok-2","supports_web_search":true,"supports_reasoning":false,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"perplexity/xai/grok-2-vision-1212":{"mode":"responses","base_model":"grok-2-vision","supports_web_search":true,"supports_reasoning":false,"provider":"perplexity","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/aisingapore/Qwen-SEA-LION-v4-32B-IT":{"mode":"chat","base_model":"qwen-sea-lion-v4-32b-it","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/allenai/Olmo-3-7B-Think":{"mode":"chat","base_model":"olmo-3-7b-think","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"publicai/allenai/Olmo-3-32B-Think":{"mode":"chat","base_model":"olmo-3-32b-think","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"source":"https://platform.publicai.co/docs","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"provider":"publicai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"qwen.qwen3-coder-480b-a35b-v1:0":{"mode":"chat","base_model":"qwen3-coder-480b-a35b","max_input_tokens":262000,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_native_structured_output":true,"source":"https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-west-2/index.json","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"qwen.qwen3-235b-a22b-2507-v1:0":{"mode":"chat","base_model":"qwen3-235b-a22b","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_native_structured_output":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"qwen.qwen3-coder-30b-a3b-v1:0":{"mode":"chat","base_model":"qwen3-coder-30b-a3b","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_native_structured_output":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"qwen.qwen3-32b-v1:0":{"mode":"chat","base_model":"qwen3-32b","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_native_structured_output":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"qwen.qwen3-next-80b-a3b":{"mode":"chat","base_model":"qwen3-next-80b-a3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"qwen.qwen3-vl-235b-a22b":{"mode":"chat","base_model":"qwen3-vl-235b-a22b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_system_messages":true,"supports_vision":true,"supports_native_structured_output":true,"supports_response_schema":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"recraft/recraftv2":{"mode":"image_generation","base_model":"recraftv2","source":"https://www.recraft.ai/docs#pricing","supported_endpoints":["/v1/images/generations"],"provider":"recraft","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"recraft/recraftv3":{"mode":"image_generation","base_model":"recraftv3","source":"https://www.recraft.ai/docs#pricing","supported_endpoints":["/v1/images/generations"],"provider":"recraft","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/meta/llama-2-13b":{"mode":"chat","base_model":"llama-2-13b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/meta/llama-2-13b-chat":{"mode":"chat","base_model":"llama-2-13b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"replicate/meta/llama-2-70b":{"mode":"chat","base_model":"llama-2-70b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/meta/llama-2-70b-chat":{"mode":"chat","base_model":"llama-2-70b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"replicate/meta/llama-2-7b":{"mode":"chat","base_model":"llama-2-7b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/meta/llama-2-7b-chat":{"mode":"chat","base_model":"llama-2-7b-chat","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"replicate/meta/llama-3-70b":{"mode":"chat","base_model":"llama-3-70b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/meta/llama-3-70b-instruct":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"replicate/meta/llama-3-8b":{"mode":"chat","base_model":"llama-3-8b","max_input_tokens":8086,"max_output_tokens":8086,"max_tokens":8086,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8086}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/meta/llama-3-8b-instruct":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8086,"max_output_tokens":8086,"max_tokens":8086,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8086}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/mistralai/mistral-7b-instruct-v0.2":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"replicate/mistralai/mistral-7b-v0.1":{"mode":"chat","base_model":"mistral-7b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"replicate/mistralai/mixtral-8x7b-instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"supports_tool_choice":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-5":{"mode":"chat","base_model":"gpt-5","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicateopenai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"replicate/anthropic/claude-4.5-haiku":{"mode":"chat","base_model":"claude-haiku-4-5","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"prompt_cache_min_tokens":4096,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/ibm-granite/granite-3.3-8b-instruct":{"mode":"chat","base_model":"granite-3.3-8b-instruct","supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-4o":{"mode":"chat","base_model":"gpt-4o","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_audio_input":true,"supports_audio_output":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/o4-mini":{"mode":"chat","base_model":"o4-mini","supports_reasoning":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/o1-mini":{"mode":"chat","base_model":"o1-mini","supports_reasoning":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/o1":{"mode":"chat","base_model":"o1","supports_reasoning":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-4o-mini":{"mode":"chat","base_model":"gpt-4o-mini","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/qwen/qwen3-235b-a22b-instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/anthropic/claude-4-sonnet":{"mode":"chat","base_model":"claude-sonnet-4","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"prompt_cache_min_tokens":1024,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/deepseek-ai/deepseek-v3":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/anthropic/claude-3.7-sonnet":{"mode":"chat","base_model":"claude-3-7-sonnet","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/anthropic/claude-3.5-haiku":{"mode":"chat","base_model":"claude-3-5-haiku","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/anthropic/claude-3.5-sonnet":{"mode":"chat","base_model":"claude-3-5-sonnet","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/google/gemini-3-pro":{"mode":"chat","base_model":"gemini-3-pro","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_audio_input":true,"supports_video_input":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/anthropic/claude-4.5-sonnet":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"prompt_cache_min_tokens":1024,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-4.1-nano":{"mode":"chat","base_model":"gpt-4.1-nano","supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-4.1-mini":{"mode":"chat","base_model":"gpt-4.1-mini","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/google/gemini-2.5-flash":{"mode":"chat","base_model":"gemini-2.5-flash","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_image_size":false,"supports_reasoning":true,"supports_video_input":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"replicate/deepseek-ai/deepseek-v3.1":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/xai/grok-4":{"mode":"chat","base_model":"grok-4","supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"replicate/deepseek-ai/deepseek-r1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"supports_reasoning":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"rerank-english-v2.0":{"mode":"rerank","base_model":"rerank-english","max_input_tokens":4096,"max_output_tokens":4096,"max_query_tokens":2048,"max_tokens":4096,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"rerank-english-v3.0":{"mode":"rerank","base_model":"rerank-english","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"cohere","max_query_tokens":2048,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"rerank-multilingual-v2.0":{"mode":"rerank","base_model":"rerank-multilingual","max_input_tokens":4096,"max_output_tokens":4096,"max_query_tokens":2048,"max_tokens":4096,"provider":"cohere","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"rerank-multilingual-v3.0":{"mode":"rerank","base_model":"rerank-multilingual","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"cohere","max_query_tokens":2048,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"rerank-v3.5":{"mode":"rerank","base_model":"rerank","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"cohere","max_query_tokens":2048,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nvidia_nim/nvidia/nv-rerankqa-mistral-4b-v3":{"mode":"rerank","base_model":"nv-rerankqa-mistral-4b-v3","provider":"nvidia_nim","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nvidia_nim/nvidia/llama-3_2-nv-rerankqa-1b-v2":{"mode":"rerank","base_model":"llama-3-2-nv-rerankqa-1b-v2","provider":"nvidia_nim","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"nvidia_nim/ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2":{"mode":"rerank","base_model":"ranking/nvidia/llama-3.2-nv-rerankqa-1b-v2","provider":"nvidia_nim","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sagemaker/meta-textgeneration-llama-2-13b":{"mode":"completion","base_model":"llama-2-13b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"sagemaker","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sagemaker/meta-textgeneration-llama-2-13b-f":{"mode":"chat","base_model":"llama-2-13b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"sagemaker","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sagemaker/meta-textgeneration-llama-2-70b":{"mode":"completion","base_model":"llama-2-70b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"sagemaker","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sagemaker/meta-textgeneration-llama-2-70b-b-f":{"mode":"chat","base_model":"llama-2-70b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"sagemaker","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sagemaker/meta-textgeneration-llama-2-7b":{"mode":"completion","base_model":"llama-2-7b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"sagemaker","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sagemaker/meta-textgeneration-llama-2-7b-f":{"mode":"chat","base_model":"llama-2-7b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"provider":"sagemaker","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/DeepSeek-R1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/DeepSeek-R1-Distill-Llama-70B":{"mode":"chat","base_model":"deepseek-r1-distill-llama-70b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"sambanova/DeepSeek-V3-0324":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://cloud.sambanova.ai/plans/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/Llama-4-Maverick-17B-128E-Instruct":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"metadata":{"notes":"For vision models, images are converted to 6432 input tokens and are billed at that amount"},"source":"https://cloud.sambanova.ai/plans/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"sambanova","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}]},"sambanova/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"metadata":{"notes":"For vision models, images are converted to 6432 input tokens and are billed at that amount"},"source":"https://cloud.sambanova.ai/plans/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"sambanova/Meta-Llama-3.1-405B-Instruct":{"mode":"chat","base_model":"llama-3.1-405b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://cloud.sambanova.ai/plans/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/Meta-Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://cloud.sambanova.ai/plans/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/Meta-Llama-3.2-1B-Instruct":{"mode":"chat","base_model":"llama-3.2-1b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/Meta-Llama-3.2-3B-Instruct":{"mode":"chat","base_model":"llama-3.2-3b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/Meta-Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://cloud.sambanova.ai/plans/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/Meta-Llama-Guard-3-8B":{"mode":"chat","base_model":"llama-guard-3-8b","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"sambanova/QwQ-32B":{"mode":"chat","base_model":"qwq-32b","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"sambanova/Qwen2-Audio-7B-Instruct":{"mode":"chat","base_model":"qwen2-audio-7b-instruct","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://cloud.sambanova.ai/plans/pricing","supports_audio_input":true,"provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sambanova/Qwen3-32B":{"mode":"chat","base_model":"qwen3-32b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.sambanova.ai/plans/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"sambanova/DeepSeek-V3.1":{"mode":"chat","base_model":"deepseek-v3.1","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}]},"sambanova/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/claude-3-5-sonnet":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":18000,"max_output_tokens":8192,"max_tokens":8192,"supports_computer_use":true,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/deepseek-r1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":32768,"max_output_tokens":8192,"max_tokens":8192,"supports_reasoning":true,"supports_system_messages":true,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/gemma-7b":{"mode":"chat","base_model":"gemma-7b","max_input_tokens":8000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/jamba-1.5-large":{"mode":"chat","base_model":"jamba-1.5-large","max_input_tokens":256000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/jamba-1.5-mini":{"mode":"chat","base_model":"jamba-1.5-mini","max_input_tokens":256000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/jamba-instruct":{"mode":"chat","base_model":"jamba-instruct","max_input_tokens":256000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/llama2-70b-chat":{"mode":"chat","base_model":"llama-2-70b-chat","max_input_tokens":4096,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/llama3-70b":{"mode":"chat","base_model":"llama-3-70b","max_input_tokens":8000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/llama3-8b":{"mode":"chat","base_model":"llama-3-8b","max_input_tokens":8000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/llama3.1-405b":{"mode":"chat","base_model":"llama-3.1-405b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/llama3.1-70b":{"mode":"chat","base_model":"llama-3.1-70b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/llama3.1-8b":{"mode":"chat","base_model":"llama-3.1-8b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_system_messages":true,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/llama3.2-1b":{"mode":"chat","base_model":"llama-3.2-1b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/llama3.2-3b":{"mode":"chat","base_model":"llama-3.2-3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/llama3.3-70b":{"mode":"chat","base_model":"llama-3.3-70b","max_tokens":8192,"max_input_tokens":128000,"max_output_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/mistral-7b":{"mode":"chat","base_model":"mistral-7b","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/mistral-large":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/mistral-large2":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/mixtral-8x7b":{"mode":"chat","base_model":"mixtral-8x7b","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"snowflake/reka-core":{"mode":"chat","base_model":"reka-core","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/reka-flash":{"mode":"chat","base_model":"reka-flash","max_input_tokens":100000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/snowflake-arctic":{"mode":"chat","base_model":"snowflake-arctic","max_input_tokens":4096,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/snowflake-llama-3.1-405b":{"mode":"chat","base_model":"snowflake-llama-3.1-405b","max_input_tokens":8000,"max_output_tokens":8192,"max_tokens":8192,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"snowflake/snowflake-llama-3.3-70b":{"mode":"chat","base_model":"snowflake-llama-3.3-70b","max_tokens":8192,"max_input_tokens":8000,"max_output_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"provider":"snowflake","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/sd3":{"mode":"image_generation","base_model":"sd3","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/sd3-large":{"mode":"image_generation","base_model":"sd3-large","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/sd3-large-turbo":{"mode":"image_generation","base_model":"sd3-large-turbo","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/sd3-medium":{"mode":"image_generation","base_model":"sd3-medium","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/sd3.5-large":{"mode":"image_generation","base_model":"sd3.5-large","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/sd3.5-large-turbo":{"mode":"image_generation","base_model":"sd3.5-large-turbo","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/sd3.5-medium":{"mode":"image_generation","base_model":"sd3.5-medium","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/stable-image-ultra":{"mode":"image_generation","base_model":"stable-image-ultra","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/inpaint":{"mode":"image_edit","base_model":"inpaint","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/outpaint":{"mode":"image_edit","base_model":"outpaint","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/erase":{"mode":"image_edit","base_model":"erase","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/search-and-replace":{"mode":"image_edit","base_model":"search-and-replace","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/search-and-recolor":{"mode":"image_edit","base_model":"search-and-recolor","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/remove-background":{"mode":"image_edit","base_model":"remove-background","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/replace-background-and-relight":{"mode":"image_edit","base_model":"replace-background-and-relight","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/sketch":{"mode":"image_edit","base_model":"sketch","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/structure":{"mode":"image_edit","base_model":"structure","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/style":{"mode":"image_edit","base_model":"style","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/style-transfer":{"mode":"image_edit","base_model":"style-transfer","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/fast":{"mode":"image_edit","base_model":"fast","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/conservative":{"mode":"image_edit","base_model":"conservative","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/creative":{"mode":"image_edit","base_model":"creative","supported_endpoints":["/v1/images/edits"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability/stable-image-core":{"mode":"image_generation","base_model":"stable-image-core","supported_endpoints":["/v1/images/generations"],"provider":"stability","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.sd3-5-large-v1:0":{"mode":"image_generation","base_model":"sd3-5-large","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.sd3-large-v1:0":{"mode":"image_generation","base_model":"sd3-large","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-core-v1:0":{"mode":"image_generation","base_model":"stable-image-core","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-conservative-upscale-v1:0":{"mode":"image_edit","base_model":"stable-conservative-upscale","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-creative-upscale-v1:0":{"mode":"image_edit","base_model":"stable-creative-upscale","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-fast-upscale-v1:0":{"mode":"image_edit","base_model":"stable-fast-upscale","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-outpaint-v1:0":{"mode":"image_edit","base_model":"stable-outpaint","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-control-sketch-v1:0":{"mode":"image_edit","base_model":"stable-image-control-sketch","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-control-structure-v1:0":{"mode":"image_edit","base_model":"stable-image-control-structure","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-erase-object-v1:0":{"mode":"image_edit","base_model":"stable-image-erase-object","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-inpaint-v1:0":{"mode":"image_edit","base_model":"stable-image-inpaint","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-remove-background-v1:0":{"mode":"image_edit","base_model":"stable-image-remove-background","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-search-recolor-v1:0":{"mode":"image_edit","base_model":"stable-image-search-recolor","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-search-replace-v1:0":{"mode":"image_edit","base_model":"stable-image-search-replace","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-style-guide-v1:0":{"mode":"image_edit","base_model":"stable-image-style-guide","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-style-transfer-v1:0":{"mode":"image_edit","base_model":"stable-style-transfer","max_input_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-core-v1:1":{"mode":"image_generation","base_model":"stable-image-core","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-ultra-v1:0":{"mode":"image_generation","base_model":"stable-image-ultra","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"stability.stable-image-ultra-v1:1":{"mode":"image_generation","base_model":"stable-image-ultra","max_input_tokens":77,"max_tokens":77,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":77}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1024-x-1024/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1024-x-1792/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"standard/1792-x-1024/dall-e-3":{"mode":"image_generation","base_model":"dall-e-3","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"linkup/search":{"mode":"search","base_model":"search","provider":"linkup","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"linkup/search-deep":{"mode":"search","base_model":"search-deep","provider":"linkup","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"tavily/search":{"mode":"search","base_model":"search","provider":"tavily","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"tavily/search-advanced":{"mode":"search","base_model":"search-advanced","provider":"tavily","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-bison":{"mode":"completion","base_model":"text-bison","max_input_tokens":8192,"max_output_tokens":2048,"max_tokens":2048,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-bison32k":{"mode":"completion","base_model":"text-bison32k","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-bison32k@002":{"mode":"completion","base_model":"text-bison32k","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-bison@001":{"mode":"completion","base_model":"text-bison","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-bison@002":{"mode":"completion","base_model":"text-bison","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-completion-codestral/codestral-2405":{"mode":"completion","base_model":"codestral","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://docs.mistral.ai/capabilities/code_generation/","provider":"text-completion-codestral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-completion-codestral/codestral-latest":{"mode":"completion","base_model":"codestral","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://docs.mistral.ai/capabilities/code_generation/","provider":"text-completion-codestral","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-embedding-004":{"mode":"embedding","base_model":"text-embedding-004","deprecation_date":"2026-01-14","max_input_tokens":2048,"max_tokens":2048,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models","provider":"vertex_ai","is_deprecated":true},"text-embedding-005":{"mode":"embedding","base_model":"text-embedding-005","deprecation_date":"2027-04-01","max_input_tokens":2048,"max_tokens":2048,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models","provider":"vertex_ai"},"text-embedding-3-large":{"mode":"embedding","base_model":"text-embedding-3-large","max_input_tokens":8191,"max_tokens":8191,"output_vector_size":3072,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai"},"text-embedding-3-small":{"mode":"embedding","base_model":"text-embedding-3-small","max_input_tokens":8191,"max_tokens":8191,"output_vector_size":1536,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai"},"text-embedding-ada-002":{"mode":"embedding","base_model":"text-embedding-ada-002","max_input_tokens":8191,"max_tokens":8191,"output_vector_size":1536,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai"},"text-embedding-ada-002-v2":{"mode":"embedding","base_model":"text-embedding-ada-002","max_input_tokens":8191,"max_tokens":8191,"provider":"openai"},"text-embedding-large-exp-03-07":{"mode":"embedding","base_model":"text-embedding-large","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":3072,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models","provider":"vertex_ai"},"text-embedding-preview-0409":{"mode":"embedding","base_model":"text-embedding","max_input_tokens":3072,"max_tokens":3072,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"vertex_ai"},"text-moderation-007":{"mode":"moderation","base_model":"text-moderation-007","max_input_tokens":32768,"max_output_tokens":0,"max_tokens":0,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-moderation-latest":{"mode":"moderation","base_model":"text-moderation","max_input_tokens":32768,"max_output_tokens":0,"max_tokens":0,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-moderation-stable":{"mode":"moderation","base_model":"text-moderation-stable","max_input_tokens":32768,"max_output_tokens":0,"max_tokens":0,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-multilingual-embedding-002":{"mode":"embedding","base_model":"text-multilingual-embedding-002","deprecation_date":"2027-04-01","max_input_tokens":2048,"max_tokens":2048,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models","provider":"vertex_ai"},"text-multilingual-embedding-preview-0409":{"mode":"embedding","base_model":"text-multilingual-embedding","max_input_tokens":3072,"max_tokens":3072,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai"},"text-unicorn":{"mode":"completion","base_model":"text-unicorn","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"text-unicorn@001":{"mode":"completion","base_model":"text-unicorn","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"textembedding-gecko":{"mode":"embedding","base_model":"textembedding-gecko","max_input_tokens":3072,"max_tokens":3072,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai"},"textembedding-gecko-multilingual":{"mode":"embedding","base_model":"textembedding-gecko-multilingual","max_input_tokens":3072,"max_tokens":3072,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai"},"textembedding-gecko-multilingual@001":{"mode":"embedding","base_model":"textembedding-gecko-multilingual","max_input_tokens":3072,"max_tokens":3072,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai"},"textembedding-gecko@001":{"mode":"embedding","base_model":"textembedding-gecko","max_input_tokens":3072,"max_tokens":3072,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai"},"textembedding-gecko@003":{"mode":"embedding","base_model":"textembedding-gecko","max_input_tokens":3072,"max_tokens":3072,"output_vector_size":768,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai"},"together-ai-21.1b-41b":{"mode":"chat","base_model":"together-ai-21.1b-41b","provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together-ai-4.1b-8b":{"mode":"chat","base_model":"together-ai-4.1b-8b","provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together-ai-41.1b-80b":{"mode":"chat","base_model":"together-ai-41.1b-80b","provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together-ai-8.1b-21b":{"mode":"chat","base_model":"together-ai-8.1b-21b","max_tokens":1000,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together-ai-81.1b-110b":{"mode":"chat","base_model":"together-ai-81.1b-110b","provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together-ai-embedding-151m-to-350m":{"mode":"embedding","base_model":"together-ai-embedding-151m-to-350m","provider":"together_ai"},"together-ai-embedding-up-to-150m":{"mode":"embedding","base_model":"together-ai-embedding-up-to-150m","provider":"together_ai"},"together_ai/baai/bge-base-en-v1.5":{"mode":"embedding","base_model":"bge-base-en","max_input_tokens":512,"output_vector_size":768,"provider":"together_ai"},"together_ai/BAAI/bge-base-en-v1.5":{"mode":"embedding","base_model":"bge-base-en","max_input_tokens":512,"output_vector_size":768,"provider":"together_ai"},"together-ai-up-to-4b":{"mode":"chat","base_model":"together-ai-up-to-4b","provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/Qwen/Qwen2.5-72B-Instruct-Turbo":{"mode":"chat","base_model":"qwen2.5-72b-instruct-turbo","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/Qwen/Qwen2.5-7B-Instruct-Turbo":{"mode":"chat","base_model":"qwen2.5-7b-instruct-turbo","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"max_input_tokens":32768,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/Qwen/Qwen3-235B-A22B-Instruct-2507-tput":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_input_tokens":262000,"source":"https://www.together.ai/models/qwen3-235b-a22b-instruct-2507-fp8","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/Qwen/Qwen3-235B-A22B-Thinking-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-thinking","max_input_tokens":256000,"source":"https://www.together.ai/models/qwen3-235b-a22b-thinking-2507","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/Qwen/Qwen3-235B-A22B-fp8-tput":{"mode":"chat","base_model":"qwen3-235b-a22b-fp8","max_input_tokens":40000,"source":"https://www.together.ai/models/qwen3-235b-a22b-fp8-tput","supports_function_calling":false,"supports_parallel_function_calling":false,"supports_tool_choice":false,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct-fp8","max_input_tokens":256000,"source":"https://www.together.ai/models/qwen3-coder-480b-a35b-instruct","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/deepseek-ai/DeepSeek-R1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":128000,"max_output_tokens":20480,"max_tokens":20480,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/deepseek-ai/DeepSeek-R1-0528-tput":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":128000,"source":"https://www.together.ai/models/deepseek-r1-0528-throughput","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/deepseek-ai/DeepSeek-V3":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"metadata":{"successor":"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813"},"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/deepseek-ai/DeepSeek-V3.1":{"mode":"chat","base_model":"deepseek-v3.1","max_tokens":128000,"source":"https://www.together.ai/models/deepseek-v3-1","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"together_ai/meta-llama/Llama-3.2-3B-Instruct-Turbo":{"mode":"chat","base_model":"llama-3.2-3b-instruct-turbo","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo":{"mode":"chat","base_model":"llama-3.3-70b-instruct-turbo","max_input_tokens":131072,"max_tokens":131072,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/meta-llama/Llama-3.3-70B-Instruct-Turbo-Free":{"mode":"chat","base_model":"llama-3.3-70b-instruct-turbo-free","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct-fp8","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"min_p","label":"Min P","helpText":"A number between 0 and 1 that can be used as an alternative to top_p and top-k","type":"number","default":0,"range":{"min":0,"max":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}]},"together_ai/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo":{"mode":"chat","base_model":"llama-3.1-405b-instruct-turbo","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo":{"mode":"chat","base_model":"llama-3.1-70b-instruct-turbo","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo":{"mode":"chat","base_model":"llama-3.1-8b-instruct-turbo","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/mistralai/Mistral-7B-Instruct-v0.1":{"mode":"chat","base_model":"mistral-7b-instruct","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/mistralai/Mistral-Small-24B-Instruct-2501":{"mode":"chat","base_model":"mistral-small-24b-instruct","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/mistralai/Mixtral-8x7B-Instruct-v0.1":{"mode":"chat","base_model":"mixtral-8x7b-instruct","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","type":"select","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call."},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/moonshotai/Kimi-K2-Instruct":{"mode":"chat","base_model":"kimi-k2-instruct","metadata":{"successor":"together_ai/moonshotai/Kimi-K3"},"source":"https://www.together.ai/models/kimi-k2-instruct","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_tokens":131072,"source":"https://www.together.ai/models/gpt-oss-120b","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":128000,"source":"https://www.together.ai/models/gpt-oss-20b","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/togethercomputer/CodeLlama-34b-Instruct":{"mode":"chat","base_model":"codellama-34b-instruct","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"together_ai/zai-org/GLM-4.5-Air-FP8":{"mode":"chat","base_model":"glm-4.5-air-fp8","max_input_tokens":128000,"source":"https://www.together.ai/models/glm-4-5-air","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/zai-org/GLM-4.6":{"mode":"chat","base_model":"glm-4.6","max_input_tokens":200000,"max_tokens":200000,"metadata":{"successor":"together_ai/zai-org/GLM-5.2"},"source":"https://www.together.ai/models/glm-4-6","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"together_ai","max_output_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/zai-org/GLM-4.7":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"https://www.together.ai/models/glm-4-7","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/moonshotai/Kimi-K2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://www.together.ai/models/kimi-k2-5","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_reasoning":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/moonshotai/Kimi-K2-Instruct-0905":{"mode":"chat","base_model":"kimi-k2-instruct","max_input_tokens":262144,"source":"https://www.together.ai/models/kimi-k2-0905","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/Qwen/Qwen3-Next-80B-A3B-Instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_input_tokens":262144,"source":"https://www.together.ai/models/qwen3-next-80b-a3b-instruct","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"together_ai/Qwen/Qwen3-Next-80B-A3B-Thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_input_tokens":262144,"source":"https://www.together.ai/models/qwen3-next-80b-a3b-thinking","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"tts-1":{"mode":"audio_speech","base_model":"tts-1","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/audio/speech"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"tts-1-hd":{"mode":"audio_speech","base_model":"tts-1-hd","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/audio/speech"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aws_polly/standard":{"mode":"audio_speech","base_model":"standard","supported_endpoints":["/v1/audio/speech"],"source":"https://aws.amazon.com/polly/pricing/","provider":"aws_polly","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aws_polly/neural":{"mode":"audio_speech","base_model":"neural","supported_endpoints":["/v1/audio/speech"],"source":"https://aws.amazon.com/polly/pricing/","provider":"aws_polly","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aws_polly/long-form":{"mode":"audio_speech","base_model":"long-form","supported_endpoints":["/v1/audio/speech"],"source":"https://aws.amazon.com/polly/pricing/","provider":"aws_polly","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"aws_polly/generative":{"mode":"audio_speech","base_model":"generative","supported_endpoints":["/v1/audio/speech"],"source":"https://aws.amazon.com/polly/pricing/","provider":"aws_polly","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.amazon.nova-lite-v1:0":{"mode":"chat","base_model":"nova-lite","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.amazon.nova-micro-v1:0":{"mode":"chat","base_model":"nova-micro","max_input_tokens":128000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.amazon.nova-premier-v1:0":{"mode":"chat","base_model":"nova-premier","max_input_tokens":1000000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_response_schema":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.amazon.nova-pro-v1:0":{"mode":"chat","base_model":"nova-pro","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-3-5-haiku-20241022-v1:0":{"mode":"chat","base_model":"claude-3-5-haiku","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"us.anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-3-5-sonnet-20241022-v2:0":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-3-7-sonnet-20250219-v1:0":{"mode":"chat","base_model":"claude-3-7-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","deprecation_date":"2026-09-10","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","ca-central-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-3-opus-20240229-v1:0":{"mode":"chat","base_model":"claude-3-opus","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","supports_cache_point":false,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-3-sonnet-20240229-v1:0":{"mode":"chat","base_model":"claude-3-sonnet","deprecation_date":"2026-07-30","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-opus-4-1-20250805-v1:0":{"mode":"chat","base_model":"claude-opus-4-1","deprecation_date":"2027-01-08","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2"],"is_deprecated":true,"tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"au.anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"us.anthropic.claude-opus-4-20250514-v1:0":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-opus-4-5-20251101-v1:0":{"mode":"chat","base_model":"claude-opus-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_output_config":false,"bedrock_output_config_effort_ceiling":"high","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"global.anthropic.claude-opus-4-5-20251101-v1:0":{"mode":"chat","base_model":"claude-opus-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_output_config":false,"bedrock_output_config_effort_ceiling":"high","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"eu.anthropic.claude-opus-4-5-20251101-v1:0":{"mode":"chat","base_model":"claude-opus-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_output_config":false,"bedrock_output_config_effort_ceiling":"high","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"us.anthropic.claude-sonnet-4-20250514-v1:0":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-10-14","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"bedrock_converse_supports_strict_tools":false,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1"],"tool_use_system_prompt_tokens":159,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"service_tiers":["default"],"supports_adaptive_thinking":false,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"us.deepseek.r1-v1:0":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":false,"supports_reasoning":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.deepseek.v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama3-1-405b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-1-405b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama3-1-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-1-70b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama3-1-8b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-1-8b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama3-2-11b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-11b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama3-2-1b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-1b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama3-2-3b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-3b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama3-2-90b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-2-90b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama3-3-70b-instruct-v1:0":{"mode":"chat","base_model":"llama-3-3-70b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama4-maverick-17b-instruct-v1:0":{"mode":"chat","base_model":"llama-4-maverick-17b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_modalities":["text","image"],"supported_output_modalities":["text","code"],"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.meta.llama4-scout-17b-instruct-v1:0":{"mode":"chat","base_model":"llama-4-scout-17b-instruct","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_modalities":["text","image"],"supported_output_modalities":["text","code"],"supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.mistral.pixtral-large-2502-v1:0":{"mode":"chat","base_model":"pixtral-large","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_tool_choice":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"v0/v0-1.0-md":{"mode":"chat","base_model":"v0-1.0-md","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"v0","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"v0/v0-1.5-lg":{"mode":"chat","base_model":"v0-1.5-lg","max_input_tokens":512000,"max_output_tokens":512000,"max_tokens":512000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"v0","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":512000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"v0/v0-1.5-md":{"mode":"chat","base_model":"v0-1.5-md","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"v0","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/alibaba/qwen-3-14b":{"mode":"chat","base_model":"qwen3-14b","max_input_tokens":40960,"max_output_tokens":16384,"max_tokens":16384,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/alibaba/qwen-3-235b":{"mode":"chat","base_model":"qwen3-235b","max_input_tokens":40960,"max_output_tokens":16384,"max_tokens":16384,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/alibaba/qwen-3-30b":{"mode":"chat","base_model":"qwen3-30b","max_input_tokens":40960,"max_output_tokens":16384,"max_tokens":16384,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/alibaba/qwen-3-32b":{"mode":"chat","base_model":"qwen3-32b","max_input_tokens":40960,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/alibaba/qwen3-coder":{"mode":"chat","base_model":"qwen3-coder","max_input_tokens":262144,"max_output_tokens":66536,"max_tokens":66536,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":66536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/amazon/nova-lite":{"mode":"chat","base_model":"nova-lite","max_input_tokens":300000,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/amazon/nova-micro":{"mode":"chat","base_model":"nova-micro","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/amazon/nova-pro":{"mode":"chat","base_model":"nova-pro","max_input_tokens":300000,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/amazon/titan-embed-text-v2":{"mode":"chat","base_model":"titan-embed-text-v2","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-3-haiku":{"mode":"chat","base_model":"claude-3-haiku","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-3-opus":{"mode":"chat","base_model":"claude-3-opus","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-3.5-haiku":{"mode":"chat","base_model":"claude-3-5-haiku","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-3.5-sonnet":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-3.7-sonnet":{"mode":"chat","base_model":"claude-3-7-sonnet","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-4-opus":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-4-sonnet":{"mode":"chat","base_model":"claude-sonnet-4","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-3-5-sonnet":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-3-5-sonnet-20241022":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-3-7-sonnet":{"mode":"chat","base_model":"claude-3-7-sonnet","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-haiku-4.5":{"mode":"chat","base_model":"claude-haiku-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-opus-4":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-opus-4.1":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-opus-4.5":{"mode":"chat","base_model":"claude-opus-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-opus-4.6":{"mode":"chat","base_model":"claude-opus-4-6","supports_adaptive_thinking":true,"supports_legacy_thinking":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-sonnet-4":{"mode":"chat","base_model":"claude-sonnet-4","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/anthropic/claude-sonnet-4.5":{"mode":"chat","base_model":"claude-sonnet-4-5","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/cohere/command-a":{"mode":"chat","base_model":"command-a","max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/cohere/command-r":{"mode":"chat","base_model":"command-r","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/cohere/command-r-plus":{"mode":"chat","base_model":"command-r-plus","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/cohere/embed-v4.0":{"mode":"chat","base_model":"embed","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/deepseek/deepseek-r1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/deepseek/deepseek-r1-distill-llama-70b":{"mode":"chat","base_model":"deepseek-r1-distill-llama-70b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/deepseek/deepseek-v3":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/google/gemini-2.0-flash":{"mode":"chat","base_model":"gemini-2.0-flash","deprecation_date":"2026-03-31","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"vercel_ai_gateway/google/gemini-2.0-flash-lite":{"mode":"chat","base_model":"gemini-2.0-flash-lite","deprecation_date":"2026-03-31","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"vercel_ai_gateway/google/gemini-2.5-flash":{"mode":"chat","base_model":"gemini-2.5-flash","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_image_size":false,"supports_reasoning":true,"supports_pdf_input":true,"supports_web_search":true,"supports_prompt_caching":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/google/gemini-2.5-pro":{"mode":"chat","base_model":"gemini-2.5-pro","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_pdf_input":true,"supports_web_search":true,"supports_prompt_caching":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/google/gemini-embedding-001":{"mode":"embedding","base_model":"gemini-embedding-001","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway"},"vercel_ai_gateway/google/gemma-2-9b":{"mode":"chat","base_model":"gemma-2-9b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/google/text-embedding-005":{"mode":"embedding","base_model":"text-embedding-005","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway"},"vercel_ai_gateway/google/text-multilingual-embedding-002":{"mode":"embedding","base_model":"text-multilingual-embedding-002","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway"},"vercel_ai_gateway/inception/mercury-coder-small":{"mode":"chat","base_model":"mercury-coder-small","max_input_tokens":32000,"max_output_tokens":16384,"max_tokens":16384,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/meta/llama-3-70b":{"mode":"chat","base_model":"llama-3-70b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/meta/llama-3-8b":{"mode":"chat","base_model":"llama-3-8b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/meta/llama-3.1-70b":{"mode":"chat","base_model":"llama-3.1-70b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/meta/llama-3.1-8b":{"mode":"chat","base_model":"llama-3.1-8b","max_input_tokens":131000,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/meta/llama-3.2-11b":{"mode":"chat","base_model":"llama-3.2-11b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/meta/llama-3.2-1b":{"mode":"chat","base_model":"llama-3.2-1b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/meta/llama-3.2-3b":{"mode":"chat","base_model":"llama-3.2-3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/meta/llama-3.2-90b":{"mode":"chat","base_model":"llama-3.2-90b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/meta/llama-3.3-70b":{"mode":"chat","base_model":"llama-3.3-70b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/meta/llama-4-maverick":{"mode":"chat","base_model":"llama-4-maverick","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/meta/llama-4-scout":{"mode":"chat","base_model":"llama-4-scout","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/codestral":{"mode":"chat","base_model":"codestral","max_input_tokens":256000,"max_output_tokens":4000,"max_tokens":4000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/codestral-embed":{"mode":"chat","base_model":"codestral-embed","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/devstral-small":{"mode":"chat","base_model":"devstral-small","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/magistral-medium":{"mode":"chat","base_model":"magistral-medium","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/magistral-small":{"mode":"chat","base_model":"magistral-small","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/ministral-3b":{"mode":"chat","base_model":"ministral-3b","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/ministral-8b":{"mode":"chat","base_model":"ministral-8b","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/mistral-embed":{"mode":"chat","base_model":"mistral-embed","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":0}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/mistral-large":{"mode":"chat","base_model":"mistral-large","max_input_tokens":32000,"max_output_tokens":4000,"max_tokens":4000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/mistral-saba-24b":{"mode":"chat","base_model":"mistral-saba-24b","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/mistral-small":{"mode":"chat","base_model":"mistral-small","max_input_tokens":32000,"max_output_tokens":4000,"max_tokens":4000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/mixtral-8x22b-instruct":{"mode":"chat","base_model":"mixtral-8x22b-instruct","max_input_tokens":65536,"max_output_tokens":2048,"max_tokens":2048,"supports_function_calling":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":2048}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vercel_ai_gateway/mistral/pixtral-12b":{"mode":"chat","base_model":"pixtral-12b","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/mistral/pixtral-large":{"mode":"chat","base_model":"pixtral-large","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/moonshotai/kimi-k2":{"mode":"chat","base_model":"kimi-k2","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/morph/morph-v3-fast":{"mode":"chat","base_model":"morph-v3-fast","max_input_tokens":32768,"max_output_tokens":16384,"max_tokens":16384,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/morph/morph-v3-large":{"mode":"chat","base_model":"morph-v3-large","max_input_tokens":32768,"max_output_tokens":16384,"max_tokens":16384,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/gpt-3.5-turbo":{"mode":"chat","base_model":"gpt-3.5-turbo","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/gpt-3.5-turbo-instruct":{"mode":"chat","base_model":"gpt-3.5-turbo-instruct","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/gpt-4-turbo":{"mode":"chat","base_model":"gpt-4-turbo","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/gpt-4.1-mini":{"mode":"chat","base_model":"gpt-4.1-mini","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/gpt-4.1-nano":{"mode":"chat","base_model":"gpt-4.1-nano","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/gpt-4o":{"mode":"chat","base_model":"gpt-4o","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/gpt-4o-mini":{"mode":"chat","base_model":"gpt-4o-mini","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/o1":{"mode":"chat","base_model":"o1","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/o3":{"mode":"chat","base_model":"o3","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/o3-mini":{"mode":"chat","base_model":"o3-mini","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/o4-mini":{"mode":"chat","base_model":"o4-mini","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":100000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/openai/text-embedding-3-large":{"mode":"embedding","base_model":"text-embedding-3-large","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway"},"vercel_ai_gateway/openai/text-embedding-3-small":{"mode":"embedding","base_model":"text-embedding-3-small","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway"},"vercel_ai_gateway/openai/text-embedding-ada-002":{"mode":"embedding","base_model":"text-embedding-ada-002","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"provider":"vercel_ai_gateway"},"vercel_ai_gateway/perplexity/sonar":{"mode":"chat","base_model":"sonar","max_input_tokens":127000,"max_output_tokens":8000,"max_tokens":8000,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/perplexity/sonar-pro":{"mode":"chat","base_model":"sonar-pro","max_input_tokens":200000,"max_output_tokens":8000,"max_tokens":8000,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/perplexity/sonar-reasoning":{"mode":"chat","base_model":"sonar","max_input_tokens":127000,"max_output_tokens":8000,"max_tokens":8000,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/perplexity/sonar-reasoning-pro":{"mode":"chat","base_model":"sonar-reasoning-pro","max_input_tokens":127000,"max_output_tokens":8000,"max_tokens":8000,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/vercel/v0-1.0-md":{"mode":"chat","base_model":"v0-1.0-md","max_input_tokens":128000,"max_output_tokens":32000,"max_tokens":32000,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/vercel/v0-1.5-md":{"mode":"chat","base_model":"v0-1.5-md","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/xai/grok-2":{"mode":"chat","base_model":"grok-2","max_input_tokens":131072,"max_output_tokens":4000,"max_tokens":4000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/xai/grok-2-vision":{"mode":"chat","base_model":"grok-2-vision","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_vision":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/xai/grok-3":{"mode":"chat","base_model":"grok-3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/xai/grok-3-fast":{"mode":"chat","base_model":"grok-3-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/xai/grok-3-mini":{"mode":"chat","base_model":"grok-3-mini","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/xai/grok-3-mini-fast":{"mode":"chat","base_model":"grok-3-mini-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/xai/grok-4":{"mode":"chat","base_model":"grok-4","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/zai/glm-4.5":{"mode":"chat","base_model":"glm-4.5","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/zai/glm-4.5-air":{"mode":"chat","base_model":"glm-4.5-air","max_input_tokens":128000,"max_output_tokens":96000,"max_tokens":96000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vercel_ai_gateway/zai/glm-4.6":{"mode":"chat","base_model":"glm-4.6","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"https://vercel.com/ai-gateway/models/glm-4.6","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"provider":"vercel_ai_gateway","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/chirp":{"mode":"audio_speech","base_model":"chirp","source":"https://cloud.google.com/text-to-speech/pricing","supported_endpoints":["/v1/audio/speech"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-5-haiku":{"mode":"chat","base_model":"claude-3-5-haiku","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-5-haiku@20241022":{"mode":"chat","base_model":"claude-3-5-haiku","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-haiku-4-5@20251001":{"mode":"chat","base_model":"claude-haiku-4-5","deprecation_date":"2026-10-15","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"regional_endpoint_uplift_multiplier":1.1,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude/haiku-4-5","supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_native_streaming":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-5-sonnet":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-5-sonnet-v2":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-5-sonnet-v2@20241022":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-5-sonnet@20240620":{"mode":"chat","base_model":"claude-3-5-sonnet","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-7-sonnet@20250219":{"mode":"chat","base_model":"claude-3-7-sonnet","deprecation_date":"2025-06-01","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"vertex_ai/claude-3-haiku":{"mode":"chat","base_model":"claude-3-haiku","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-haiku@20240307":{"mode":"chat","base_model":"claude-3-haiku","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-opus":{"mode":"chat","base_model":"claude-3-opus","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-opus@20240229":{"mode":"chat","base_model":"claude-3-opus","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-sonnet":{"mode":"chat","base_model":"claude-3-sonnet","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-3-sonnet@20240229":{"mode":"chat","base_model":"claude-3-sonnet","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-opus-4":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-opus-4-1":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-opus-4-1@20250805":{"mode":"chat","base_model":"claude-opus-4-1","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-opus-4-5":{"mode":"chat","base_model":"claude-opus-4-5","deprecation_date":"2026-11-24","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"regional_endpoint_uplift_multiplier":1.1,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai-anthropic_models","supports_web_search":true,"tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-opus-4-5@20251101":{"mode":"chat","base_model":"claude-opus-4-5","deprecation_date":"2026-11-24","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"regional_endpoint_uplift_multiplier":1.1,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_streaming":true,"supports_output_config":true,"prompt_cache_min_tokens":4096,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai-anthropic_models","supports_web_search":true,"tool_use_system_prompt_tokens":159,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-opus-4-6":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"vertex_ai-anthropic_models","supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-sonnet-4-5":{"mode":"chat","base_model":"claude-sonnet-4-5","deprecation_date":"2026-09-29","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"regional_endpoint_uplift_multiplier":1.1,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai-anthropic_models","supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-sonnet-4-5@20250929":{"mode":"chat","base_model":"claude-sonnet-4-5","deprecation_date":"2026-09-29","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"regional_endpoint_uplift_multiplier":1.1,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_streaming":true,"prompt_cache_min_tokens":1024,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-opus-4@20250514":{"mode":"chat","base_model":"claude-opus-4","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-sonnet-4":{"mode":"chat","base_model":"claude-sonnet-4","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-sonnet-4@20250514":{"mode":"chat","base_model":"claude-sonnet-4","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":159,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistralai/codestral-2@001":{"mode":"chat","base_model":"codestral-2","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/codestral-2":{"mode":"chat","base_model":"codestral-2","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/codestral-2@001":{"mode":"chat","base_model":"codestral-2","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistralai/codestral-2":{"mode":"chat","base_model":"codestral-2","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/codestral-2501":{"mode":"chat","base_model":"codestral","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/codestral@2405":{"mode":"chat","base_model":"codestral","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/codestral@latest":{"mode":"chat","base_model":"codestral","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/deepseek-ai/deepseek-v3.1-maas":{"mode":"chat","base_model":"deepseek-v3.1","max_input_tokens":163840,"max_output_tokens":32768,"max_tokens":32768,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supported_regions":["us-west2"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}]},"vertex_ai/deepseek-ai/deepseek-v3.2-maas":{"mode":"chat","base_model":"deepseek-v3.2","max_input_tokens":163840,"max_output_tokens":32768,"max_tokens":32768,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supported_regions":["us-west2"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/deepseek-ai/deepseek-r1-0528-maas":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":65336,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supported_regions":["us-central1"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/gemini-2.5-flash-image":{"mode":"image_generation","base_model":"gemini-2.5-flash-image","deprecation_date":"2026-10-02","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"rpm":100000,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/multimodal/image-generation#edit-an-image","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","image"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":false,"tpm":8000000,"supports_image_size":false,"provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/gemini-3-pro-image-preview":{"mode":"image_generation","base_model":"gemini-3-pro-image","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/3-pro-image","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/deep-research-pro-preview-12-2025":{"mode":"image_generation","base_model":"deep-research-pro","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.cloud.google.com/vertex-ai/generative-ai/docs/models/gemini/3-pro-image","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/imagegeneration@006":{"mode":"image_generation","base_model":"imagegeneration","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/imagen-3.0-fast-generate-001":{"mode":"image_generation","base_model":"imagen-3.0-fast-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/imagen-3.0-generate-001":{"mode":"image_generation","base_model":"imagen-3.0-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/imagen-3.0-generate-002":{"mode":"image_generation","base_model":"imagen-3.0-generate","deprecation_date":"2025-11-10","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"vertex_ai/imagen-3.0-capability-001":{"mode":"image_generation","base_model":"imagen-3.0-capability","source":"https://cloud.google.com/vertex-ai/generative-ai/docs/image/edit-insert-objects","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/imagen-4.0-fast-generate-001":{"mode":"image_generation","base_model":"imagen-4.0-fast-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/imagen-4.0-generate-001":{"mode":"image_generation","base_model":"imagen-4.0-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/imagen-4.0-ultra-generate-001":{"mode":"image_generation","base_model":"imagen-4.0-ultra-generate","source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/jamba-1.5":{"mode":"chat","base_model":"jamba-1.5","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/jamba-1.5-large":{"mode":"chat","base_model":"jamba-1.5-large","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/jamba-1.5-large@001":{"mode":"chat","base_model":"jamba-1.5-large","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/jamba-1.5-mini":{"mode":"chat","base_model":"jamba-1.5-mini","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/jamba-1.5-mini@001":{"mode":"chat","base_model":"jamba-1.5-mini","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/meta/llama-3.1-405b-instruct-maas":{"mode":"chat","base_model":"llama-3.1-405b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"source":"https://console.cloud.google.com/vertex-ai/publishers/meta/model-garden/llama-3.2-90b-vision-instruct-maas","supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/meta/llama-3.1-70b-instruct-maas":{"mode":"chat","base_model":"llama-3.1-70b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"source":"https://console.cloud.google.com/vertex-ai/publishers/meta/model-garden/llama-3.2-90b-vision-instruct-maas","supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/meta/llama-3.1-8b-instruct-maas":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"metadata":{"notes":"VertexAI states that The Llama 3.1 API service for llama-3.1-70b-instruct-maas and llama-3.1-8b-instruct-maas are in public preview and at no cost."},"source":"https://console.cloud.google.com/vertex-ai/publishers/meta/model-garden/llama-3.2-90b-vision-instruct-maas","supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/meta/llama-3.2-90b-vision-instruct-maas":{"mode":"chat","base_model":"llama-3.2-90b-vision-instruct","max_input_tokens":128000,"max_output_tokens":2048,"max_tokens":2048,"metadata":{"notes":"VertexAI states that The Llama 3.2 API service is at no cost during public preview, and will be priced as per dollar-per-1M-tokens at GA."},"source":"https://console.cloud.google.com/vertex-ai/publishers/meta/model-garden/llama-3.2-90b-vision-instruct-maas","supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/meta/llama-4-maverick-17b-128e-instruct-maas":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supported_modalities":["text","image"],"supported_output_modalities":["text","code"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}]},"vertex_ai/meta/llama-4-maverick-17b-16e-instruct-maas":{"mode":"chat","base_model":"llama-4-maverick-17b-16e-instruct","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supported_modalities":["text","image"],"supported_output_modalities":["text","code"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/meta/llama-4-scout-17b-128e-instruct-maas":{"mode":"chat","base_model":"llama-4-scout-17b-128e-instruct","max_input_tokens":10000000,"max_output_tokens":10000000,"max_tokens":10000000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supported_modalities":["text","image"],"supported_output_modalities":["text","code"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":10000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/meta/llama-4-scout-17b-16e-instruct-maas":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_input_tokens":10000000,"max_output_tokens":10000000,"max_tokens":10000000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supported_modalities":["text","image"],"supported_output_modalities":["text","code"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":10000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vertex_ai/meta/llama3-405b-instruct-maas":{"mode":"chat","base_model":"llama-3-405b-instruct","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/meta/llama3-70b-instruct-maas":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vertex_ai/meta/llama3-8b-instruct-maas":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/minimaxai/minimax-m2-maas":{"mode":"chat","base_model":"minimax-m2","max_input_tokens":196608,"max_output_tokens":196608,"max_tokens":196608,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":196608}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/moonshotai/kimi-k2-thinking-maas":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/zai-org/glm-4.7-maas":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#partner-models","supported_regions":["global"],"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/zai-org/glm-5-maas":{"mode":"chat","base_model":"glm-5","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#glm-models","supported_regions":["global"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-medium-3":{"mode":"chat","base_model":"mistral-medium-3","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-medium-3@001":{"mode":"chat","base_model":"mistral-medium-3","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistralai/mistral-medium-3":{"mode":"chat","base_model":"mistral-medium-3","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistralai/mistral-medium-3@001":{"mode":"chat","base_model":"mistral-medium-3","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-large-2411":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-large@2407":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-large@2411-001":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-large@latest":{"mode":"chat","base_model":"mistral-large","max_input_tokens":128000,"max_output_tokens":8191,"max_tokens":8191,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-nemo@2407":{"mode":"chat","base_model":"mistral-nemo","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-nemo@latest":{"mode":"chat","base_model":"mistral-nemo","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-small-2503":{"mode":"chat","base_model":"mistral-small","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-small-2503@001":{"mode":"chat","base_model":"mistral-small","max_input_tokens":32000,"max_output_tokens":8191,"max_tokens":8191,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/mistral-ocr-2505":{"mode":"ocr","base_model":"mistral-ocr","supported_endpoints":["/v1/ocr"],"source":"https://cloud.google.com/generative-ai-app-builder/pricing","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/deepseek-ai/deepseek-ocr-maas":{"mode":"ocr","base_model":"deepseek-ocr","source":"https://cloud.google.com/vertex-ai/pricing","supported_regions":["us-central1"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/openai/gpt-oss-120b-maas":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://console.cloud.google.com/vertex-ai/publishers/openai/model-garden/gpt-oss-120b-maas","supports_reasoning":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vertex_ai/openai/gpt-oss-20b-maas":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://console.cloud.google.com/vertex-ai/publishers/openai/model-garden/gpt-oss-120b-maas","supports_reasoning":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"vertex_ai/qwen/qwen3-235b-a22b-instruct-2507-maas":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_input_tokens":262144,"max_output_tokens":16384,"max_tokens":16384,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_regions":["global"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/qwen/qwen3-coder-480b-a35b-instruct-maas":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_regions":["global"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/qwen/qwen3-next-80b-a3b-instruct-maas":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_regions":["global"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/qwen/qwen3-next-80b-a3b-thinking-maas":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_regions":["global"],"supports_function_calling":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/veo-2.0-generate-001":{"mode":"video_generation","base_model":"veo-2.0-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/veo-3.0-fast-generate-preview":{"mode":"video_generation","base_model":"veo-3.0-fast-generate","deprecation_date":"2025-11-12","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"vertex_ai/veo-3.0-generate-preview":{"mode":"video_generation","base_model":"veo-3.0-generate","deprecation_date":"2025-11-12","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"is_deprecated":true},"vertex_ai/veo-3.0-fast-generate-001":{"mode":"video_generation","base_model":"veo-3.0-fast-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/veo-3.0-generate-001":{"mode":"video_generation","base_model":"veo-3.0-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/veo-3.1-generate-preview":{"mode":"video_generation","base_model":"veo-3.1-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/veo-3.1-fast-generate-preview":{"mode":"video_generation","base_model":"veo-3.1-fast-generate","max_input_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/veo-3.1-generate-001":{"mode":"video_generation","base_model":"veo-3.1-generate","deprecation_date":"2026-11-17","max_input_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/veo-3.1-fast-generate-001":{"mode":"video_generation","base_model":"veo-3.1-fast-generate","deprecation_date":"2026-11-17","max_input_tokens":1024,"max_tokens":1024,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"voyage/rerank-2":{"mode":"rerank","base_model":"rerank-2","max_input_tokens":16000,"max_output_tokens":16000,"max_tokens":16000,"provider":"voyage","max_query_tokens":16000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"voyage/rerank-2-lite":{"mode":"rerank","base_model":"rerank-2-lite","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"provider":"voyage","max_query_tokens":8000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"voyage/rerank-2.5":{"mode":"rerank","base_model":"rerank-2.5","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"provider":"voyage","max_query_tokens":32000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"voyage/rerank-2.5-lite":{"mode":"rerank","base_model":"rerank-2.5-lite","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"provider":"voyage","max_query_tokens":32000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"voyage/voyage-2":{"mode":"embedding","base_model":"voyage-2","max_input_tokens":4000,"max_tokens":4000,"provider":"voyage"},"voyage/voyage-3":{"mode":"embedding","base_model":"voyage-3","max_input_tokens":32000,"max_tokens":32000,"provider":"voyage"},"voyage/voyage-3-large":{"mode":"embedding","base_model":"voyage-3-large","max_input_tokens":32000,"max_tokens":32000,"provider":"voyage"},"voyage/voyage-3-lite":{"mode":"embedding","base_model":"voyage-3-lite","max_input_tokens":32000,"max_tokens":32000,"provider":"voyage"},"voyage/voyage-3.5":{"mode":"embedding","base_model":"voyage-3.5","max_input_tokens":32000,"max_tokens":32000,"provider":"voyage"},"voyage/voyage-3.5-lite":{"mode":"embedding","base_model":"voyage-3.5-lite","max_input_tokens":32000,"max_tokens":32000,"provider":"voyage"},"voyage/voyage-code-2":{"mode":"embedding","base_model":"voyage-code-2","max_input_tokens":16000,"max_tokens":16000,"provider":"voyage"},"voyage/voyage-code-3":{"mode":"embedding","base_model":"voyage-code-3","max_input_tokens":32000,"max_tokens":32000,"provider":"voyage"},"voyage/voyage-context-3":{"mode":"embedding","base_model":"voyage-context-3","max_input_tokens":120000,"max_tokens":120000,"provider":"voyage"},"voyage/voyage-finance-2":{"mode":"embedding","base_model":"voyage-finance-2","max_input_tokens":32000,"max_tokens":32000,"provider":"voyage"},"voyage/voyage-large-2":{"mode":"embedding","base_model":"voyage-large-2","max_input_tokens":16000,"max_tokens":16000,"provider":"voyage"},"voyage/voyage-law-2":{"mode":"embedding","base_model":"voyage-law-2","max_input_tokens":16000,"max_tokens":16000,"provider":"voyage"},"voyage/voyage-lite-01":{"mode":"embedding","base_model":"voyage-lite-01","max_input_tokens":4096,"max_tokens":4096,"provider":"voyage"},"voyage/voyage-lite-02-instruct":{"mode":"embedding","base_model":"voyage-lite-02-instruct","max_input_tokens":4000,"max_tokens":4000,"provider":"voyage"},"voyage/voyage-multimodal-3":{"mode":"embedding","base_model":"voyage-multimodal-3","max_input_tokens":32000,"max_tokens":32000,"provider":"voyage"},"wandb/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","supports_reasoning":true,"max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"wandb/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","supports_reasoning":true,"max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"wandb/zai-org/GLM-4.5":{"mode":"chat","base_model":"glm-4.5","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/Qwen/Qwen3-235B-A22B-Instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/Qwen/Qwen3-Coder-480B-A35B-Instruct":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/Qwen/Qwen3-235B-A22B-Thinking-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/moonshotai/Kimi-K2-Instruct":{"mode":"chat","base_model":"kimi-k2-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/meta-llama/Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/deepseek-ai/DeepSeek-V3.1":{"mode":"chat","base_model":"deepseek-v3.1","supports_reasoning":true,"max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}]},"wandb/deepseek-ai/DeepSeek-R1-0528":{"mode":"chat","base_model":"deepseek-r1","max_tokens":161000,"max_input_tokens":161000,"max_output_tokens":161000,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":161000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/deepseek-ai/DeepSeek-V3-0324":{"mode":"chat","base_model":"deepseek-v3","max_tokens":161000,"max_input_tokens":161000,"max_output_tokens":161000,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":161000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"wandb/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_tokens":64000,"max_input_tokens":64000,"max_output_tokens":64000,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"wandb/microsoft/Phi-4-mini-instruct":{"mode":"chat","base_model":"phi-4-mini-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"wandb","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-3-8b-instruct":{"mode":"chat","base_model":"granite-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":1024,"max_tokens":1024,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":1024}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/mistralai/mistral-large":{"mode":"chat","base_model":"mistral-large","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/bigscience/mt0-xxl-13b":{"mode":"chat","base_model":"mt0-xxl-13b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"source":"https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models.html?context=wx","provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/core42/jais-13b-chat":{"mode":"chat","base_model":"jais-13b-chat","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/google/flan-t5-xl-3b":{"mode":"chat","base_model":"flan-t5-xl-3b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-13b-chat-v2":{"mode":"chat","base_model":"granite-13b-chat-v2","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-13b-instruct-v2":{"mode":"chat","base_model":"granite-13b-instruct-v2","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-3-3-8b-instruct":{"mode":"chat","base_model":"granite-3-3-8b-instruct","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-4-h-small":{"mode":"chat","base_model":"granite-4-h-small","max_tokens":20480,"max_input_tokens":20480,"max_output_tokens":20480,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"source":"https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models.html?context=wx","provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":20480}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-guardian-3-2-2b":{"mode":"chat","base_model":"granite-guardian-3-2-2b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-guardian-3-3-8b":{"mode":"chat","base_model":"granite-guardian-3-3-8b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-ttm-1024-96-r2":{"mode":"chat","base_model":"granite-ttm-1024-96-r2","max_tokens":512,"max_input_tokens":512,"max_output_tokens":512,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-ttm-1536-96-r2":{"mode":"chat","base_model":"granite-ttm-1536-96-r2","max_tokens":512,"max_input_tokens":512,"max_output_tokens":512,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-ttm-512-96-r2":{"mode":"chat","base_model":"granite-ttm-512-96-r2","max_tokens":512,"max_input_tokens":512,"max_output_tokens":512,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/ibm/granite-vision-3-2-2b":{"mode":"chat","base_model":"granite-vision-3-2-2b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":true,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/meta-llama/llama-3-2-11b-vision-instruct":{"mode":"chat","base_model":"llama-3-2-11b-vision-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/meta-llama/llama-3-2-1b-instruct":{"mode":"chat","base_model":"llama-3-2-1b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/meta-llama/llama-3-2-3b-instruct":{"mode":"chat","base_model":"llama-3-2-3b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/meta-llama/llama-3-2-90b-vision-instruct":{"mode":"chat","base_model":"llama-3-2-90b-vision-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/meta-llama/llama-3-3-70b-instruct":{"mode":"chat","base_model":"llama-3-3-70b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"source":"https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models.html?context=wx","provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/meta-llama/llama-4-maverick-17b":{"mode":"chat","base_model":"llama-4-maverick-17b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"source":"https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models.html?context=wx","provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/meta-llama/llama-guard-3-11b-vision":{"mode":"chat","base_model":"llama-guard-3-11b-vision","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":true,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/mistralai/mistral-medium-2505":{"mode":"chat","base_model":"mistral-medium","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/mistralai/mistral-small-2503":{"mode":"chat","base_model":"mistral-small","max_tokens":32000,"max_input_tokens":32000,"max_output_tokens":32000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/mistralai/mistral-small-3-1-24b-instruct-2503":{"mode":"chat","base_model":"mistral-small-3-1-24b-instruct","max_tokens":32000,"max_input_tokens":32000,"max_output_tokens":32000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"source":"https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models.html?context=wx","provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/mistralai/pixtral-12b-2409":{"mode":"chat","base_model":"pixtral-12b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":true,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"source":"https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models.html?context=wx","provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"watsonx/sdaia/allam-1-13b-instruct":{"mode":"chat","base_model":"allam-1-13b-instruct","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"watsonx/whisper-large-v3-turbo":{"mode":"audio_transcription","base_model":"whisper-large-v3-turbo","supported_endpoints":["/v1/audio/transcriptions"],"provider":"watsonx","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"whisper-1":{"mode":"audio_transcription","base_model":"whisper-1","supported_endpoints":["/v1/audio/transcriptions"],"deprecation_date":"2027-02-26","source":"https://developers.openai.com/api/docs/pricing","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-2":{"mode":"chat","base_model":"grok-2","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"xai/grok-2-1212":{"mode":"chat","base_model":"grok-2","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-2-latest":{"mode":"chat","base_model":"grok-2","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"xai/grok-2-vision":{"mode":"chat","base_model":"grok-2-vision","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"xai/grok-2-vision-1212":{"mode":"chat","base_model":"grok-2-vision","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-2-vision-latest":{"mode":"chat","base_model":"grok-2-vision","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"xai/grok-3":{"mode":"chat","base_model":"grok-3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-beta":{"mode":"chat","base_model":"grok-3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-fast-beta":{"mode":"chat","base_model":"grok-3-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-fast-latest":{"mode":"chat","base_model":"grok-3-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-latest":{"mode":"chat","base_model":"grok-3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-mini":{"mode":"chat","base_model":"grok-3-mini","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-mini-beta":{"mode":"chat","base_model":"grok-3-mini","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-mini-fast":{"mode":"chat","base_model":"grok-3-mini-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-mini-fast-beta":{"mode":"chat","base_model":"grok-3-mini-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-mini-fast-latest":{"mode":"chat","base_model":"grok-3-mini-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-3-mini-latest":{"mode":"chat","base_model":"grok-3-mini","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://x.ai/api#pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4":{"mode":"chat","base_model":"grok-4","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4-fast-reasoning":{"mode":"chat","base_model":"grok-4-fast","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4-fast-non-reasoning":{"mode":"chat","base_model":"grok-4-fast-non","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4-0709":{"mode":"chat","base_model":"grok-4","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"xai/grok-4-latest":{"mode":"chat","base_model":"grok-4","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4-1-fast":{"mode":"chat","base_model":"grok-4-1-fast","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","supports_audio_input":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4-1-fast-reasoning":{"mode":"chat","base_model":"grok-4-1-fast","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","supports_audio_input":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4-1-fast-reasoning-latest":{"mode":"chat","base_model":"grok-4-1-fast","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","supports_audio_input":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4-1-fast-non-reasoning":{"mode":"chat","base_model":"grok-4-1-fast-non","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning","supports_audio_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-4-1-fast-non-reasoning-latest":{"mode":"chat","base_model":"grok-4-1-fast-non","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning","supports_audio_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-beta":{"mode":"chat","base_model":"grok","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"xai/grok-code-fast":{"mode":"chat","base_model":"grok-code-fast","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"xai","supported_modalities":["text","image"],"supported_output_modalities":["text"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-code-fast-1":{"mode":"chat","base_model":"grok-code-fast-1","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"xai","supported_modalities":["text","image"],"supported_output_modalities":["text"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-code-fast-1-0825":{"mode":"chat","base_model":"grok-code-fast-1","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"xai","supported_modalities":["text","image"],"supported_output_modalities":["text"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"xai/grok-vision-beta":{"mode":"chat","base_model":"grok-vision","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"zai.glm-4.7":{"mode":"chat","base_model":"zai.glm-4.7","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4.7":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":200000,"max_output_tokens":128000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4.6":{"mode":"chat","base_model":"glm-4.6","max_input_tokens":200000,"max_output_tokens":128000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4.5":{"mode":"chat","base_model":"glm-4.5","max_input_tokens":128000,"max_output_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4.5v":{"mode":"chat","base_model":"glm-4.5v","max_input_tokens":128000,"max_output_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4.5-x":{"mode":"chat","base_model":"glm-4.5-x","max_input_tokens":128000,"max_output_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4.5-air":{"mode":"chat","base_model":"glm-4.5-air","max_input_tokens":128000,"max_output_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4.5-airx":{"mode":"chat","base_model":"glm-4.5-airx","max_input_tokens":128000,"max_output_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4-32b-0414-128k":{"mode":"chat","base_model":"glm-4-32b-0414-128k","max_input_tokens":128000,"max_output_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"zai/glm-4.5-flash":{"mode":"chat","base_model":"glm-4.5-flash","max_input_tokens":128000,"max_output_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/search_api":{"mode":"vector_store","base_model":"search-api","provider":"vertex_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai/container":{"mode":"chat","base_model":"container","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai/sora-2":{"mode":"video_generation","base_model":"sora-2","source":"https://platform.openai.com/docs/api-reference/videos","supported_modalities":["text","image"],"supported_output_modalities":["video"],"supported_resolutions":["720x1280","1280x720"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai/sora-2-pro":{"mode":"video_generation","base_model":"sora-2-pro","source":"https://platform.openai.com/docs/api-reference/videos","supported_modalities":["text","image"],"supported_output_modalities":["video"],"supported_resolutions":["720x1280","1280x720"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai/sora-2-pro-high-res":{"mode":"video_generation","base_model":"sora-2-pro","source":"https://platform.openai.com/docs/api-reference/videos","supported_modalities":["text","image"],"supported_output_modalities":["video"],"supported_resolutions":["1024x1792","1792x1024"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/sora-2":{"mode":"video_generation","base_model":"sora-2","deprecation_date":"2026-10-15","source":"https://azure.microsoft.com/en-us/products/ai-services/video-generation","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"azure","supported_resolutions":["720x1280","1280x720"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/sora-2-pro":{"mode":"video_generation","base_model":"sora-2-pro","source":"https://azure.microsoft.com/en-us/products/ai-services/video-generation","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"azure","supported_resolutions":["720x1280","1280x720"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"azure/sora-2-pro-high-res":{"mode":"video_generation","base_model":"sora-2-pro","source":"https://azure.microsoft.com/en-us/products/ai-services/video-generation","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"azure","supported_resolutions":["1024x1792","1792x1024"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"runwayml/gen4_turbo":{"mode":"video_generation","base_model":"gen4-turbo","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["video"],"metadata":{"comment":"5 credits per second @ $0.01 per credit = $0.05 per second"},"provider":"runwayml","supported_resolutions":["1280x720","720x1280"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"runwayml/gen4_aleph":{"mode":"video_generation","base_model":"gen4-aleph","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["video"],"supported_resolutions":["1280x720","720x1280"],"metadata":{"comment":"15 credits per second @ $0.01 per credit = $0.15 per second"},"provider":"runwayml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"runwayml/gen3a_turbo":{"mode":"video_generation","base_model":"gen3a-turbo","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["video"],"supported_resolutions":["1280x720","720x1280"],"metadata":{"comment":"5 credits per second @ $0.01 per credit = $0.05 per second"},"provider":"runwayml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"runwayml/gen4_image":{"mode":"image_generation","base_model":"gen4-image","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["image"],"metadata":{"comment":"5 credits per 720p image or 8 credits per 1080p image @ $0.01 per credit. Using 5 credits ($0.05) as base cost"},"provider":"runwayml","supported_resolutions":["1280x720","1920x1080"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"runwayml/gen4_image_turbo":{"mode":"image_generation","base_model":"gen4-image-turbo","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["image"],"metadata":{"comment":"2 credits per image (any resolution) @ $0.01 per credit = $0.02 per image"},"provider":"runwayml","supported_resolutions":["1280x720","1920x1080"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"runwayml/eleven_multilingual_v2":{"mode":"audio_speech","base_model":"eleven-multilingual-v2","source":"https://docs.dev.runwayml.com/guides/pricing/","metadata":{"comment":"Estimated cost based on standard TTS pricing. RunwayML uses ElevenLabs models."},"provider":"runwayml","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-coder-480b-a35b-instruct":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_reasoning":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/flux-kontext-pro":{"mode":"image_generation","base_model":"flux-kontext-pro","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/SSD-1B":{"mode":"image_generation","base_model":"ssd-1b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/chronos-hermes-13b-v2":{"mode":"chat","base_model":"chronos-hermes-13b-v2","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/code-llama-13b":{"mode":"chat","base_model":"codellama-13b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-13b-instruct":{"mode":"chat","base_model":"codellama-13b-instruct","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-13b-python":{"mode":"chat","base_model":"codellama-13b-python","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-34b":{"mode":"chat","base_model":"codellama-34b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-34b-instruct":{"mode":"chat","base_model":"codellama-34b-instruct","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-34b-python":{"mode":"chat","base_model":"codellama-34b-python","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-70b":{"mode":"chat","base_model":"codellama-70b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-70b-instruct":{"mode":"chat","base_model":"codellama-70b-instruct","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-70b-python":{"mode":"chat","base_model":"codellama-70b-python","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-7b":{"mode":"chat","base_model":"codellama-7b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-7b-instruct":{"mode":"chat","base_model":"codellama-7b-instruct","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-llama-7b-python":{"mode":"chat","base_model":"codellama-7b-python","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/code-qwen-1p5-7b":{"mode":"chat","base_model":"code-qwen-1.5-7b","max_tokens":65536,"max_input_tokens":65536,"max_output_tokens":65536,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/codegemma-2b":{"mode":"chat","base_model":"codegemma-2b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/codegemma-7b":{"mode":"chat","base_model":"codegemma-7b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/cogito-671b-v2-p1":{"mode":"chat","base_model":"cogito-671b-v2-p1","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-3b":{"mode":"chat","base_model":"cogito-v1-llama-3b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-70b":{"mode":"chat","base_model":"cogito-v1-llama-70b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/cogito-v1-preview-llama-8b":{"mode":"chat","base_model":"cogito-v1-llama-8b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/cogito-v1-preview-qwen-14b":{"mode":"chat","base_model":"cogito-v1-qwen-14b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/cogito-v1-preview-qwen-32b":{"mode":"chat","base_model":"cogito-v1-qwen-32b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/flux-kontext-max":{"mode":"image_generation","base_model":"flux-kontext-max","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/dbrx-instruct":{"mode":"chat","base_model":"dbrx-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/deepseek-coder-1b-base":{"mode":"chat","base_model":"deepseek-coder-1b-base","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-coder-33b-instruct":{"mode":"chat","base_model":"deepseek-coder-33b-instruct","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-base":{"mode":"chat","base_model":"deepseek-coder-7b-base","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-base-v1p5":{"mode":"chat","base_model":"deepseek-coder-7b-base-v1.5","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-coder-7b-instruct-v1p5":{"mode":"chat","base_model":"deepseek-coder-7b-instruct-v1.5","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-lite-base":{"mode":"chat","base_model":"deepseek-coder-v2-lite-base","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-coder-v2-lite-instruct":{"mode":"chat","base_model":"deepseek-coder-v2-lite-instruct","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-prover-v2":{"mode":"chat","base_model":"deepseek-prover-v2","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-r1-0528-distill-qwen3-8b":{"mode":"chat","base_model":"deepseek-r1-0528-distill-qwen3-8b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-llama-70b":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/deepseek-r1-distill-llama-70b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-llama-8b":{"mode":"chat","base_model":"deepseek-r1-distill-llama-8b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-14b":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-14b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-1p5b":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-1.5b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-32b":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-32b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-r1-distill-qwen-7b":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-7b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-v2-lite-chat":{"mode":"chat","base_model":"deepseek-v2-lite-chat","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/deepseek-v2p5":{"mode":"chat","base_model":"deepseek-v2.5","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/devstral-small-2505":{"mode":"chat","base_model":"devstral-small","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/dobby-mini-unhinged-plus-llama-3-1-8b":{"mode":"chat","base_model":"dobby-mini-unhinged-plus-llama-3-1-8b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/dobby-unhinged-llama-3-3-70b-new":{"mode":"chat","base_model":"dobby-unhinged-llama-3-3-70b-new","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/dolphin-2-9-2-qwen2-72b":{"mode":"chat","base_model":"dolphin-2-9-2-qwen2-72b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/dolphin-2p6-mixtral-8x7b":{"mode":"chat","base_model":"dolphin-2.6-mixtral-8x7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/ernie-4p5-21b-a3b-pt":{"mode":"chat","base_model":"ernie-4.5-21b-a3b-pt","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/ernie-4p5-300b-a47b-pt":{"mode":"chat","base_model":"ernie-4.5-300b-a47b-pt","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/fare-20b":{"mode":"chat","base_model":"fare-20b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/firefunction-v1":{"mode":"chat","base_model":"firefunction-v1","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/firellava-13b":{"mode":"chat","base_model":"firellava-13b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/firesearch-ocr-v6":{"mode":"image_generation","base_model":"fireworks/accounts/fireworks/models/firesearch-ocr-v6","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/fireworks-asr-large":{"mode":"audio_transcription","base_model":"fireworks-asr-large","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/fireworks-asr-v2":{"mode":"audio_transcription","base_model":"fireworks-asr-v2","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/flux-1-dev":{"mode":"chat","base_model":"flux-1-dev","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/flux-1-dev-controlnet-union":{"mode":"chat","base_model":"flux-1-dev-controlnet-union","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/flux-1-dev-fp8":{"mode":"image_generation","base_model":"flux-1-dev-fp8","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/flux-1-schnell":{"mode":"chat","base_model":"flux-1-schnell","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/flux-1-schnell-fp8":{"mode":"image_generation","base_model":"flux-1-schnell-fp8","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/gemma-2b-it":{"mode":"chat","base_model":"gemma-2b-it","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/gemma-3-27b-it":{"mode":"chat","base_model":"gemma-3-27b-it","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/gemma-7b":{"mode":"chat","base_model":"gemma-7b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/gemma-7b-it":{"mode":"chat","base_model":"gemma-7b-it","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/gemma2-9b-it":{"mode":"chat","base_model":"gemma-2-9b-it","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/glm-4p5v":{"mode":"chat","base_model":"glm-4.5v","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_reasoning":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/gpt-oss-safeguard-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/gpt-oss-safeguard-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/hermes-2-pro-mistral-7b":{"mode":"chat","base_model":"hermes-2-pro-mistral-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/internvl3-38b":{"mode":"chat","base_model":"internvl3-38b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/internvl3-78b":{"mode":"chat","base_model":"internvl3-78b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/internvl3-8b":{"mode":"chat","base_model":"internvl3-8b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/japanese-stable-diffusion-xl":{"mode":"image_generation","base_model":"japanese-stable-diffusion-xl","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/kat-coder":{"mode":"chat","base_model":"kat-coder","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/kat-dev-32b":{"mode":"chat","base_model":"kat-dev-32b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/kat-dev-72b-exp":{"mode":"chat","base_model":"kat-dev-72b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-guard-2-8b":{"mode":"chat","base_model":"llama-guard-2-8b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-guard-3-1b":{"mode":"chat","base_model":"llama-guard-3-1b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-guard-3-8b":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/llama-guard-3-8b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/llama-v2-13b":{"mode":"chat","base_model":"llama-2-13b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v2-13b-chat":{"mode":"chat","base_model":"llama-2-13b-chat","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v2-70b":{"mode":"chat","base_model":"llama-2-70b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v2-70b-chat":{"mode":"chat","base_model":"llama-2-70b-chat","max_tokens":2048,"max_input_tokens":2048,"max_output_tokens":2048,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":2048}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v2-7b":{"mode":"chat","base_model":"llama-2-7b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v2-7b-chat":{"mode":"chat","base_model":"llama-2-7b-chat","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct":{"mode":"chat","base_model":"llama-3-70b-instruct","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v3-70b-instruct-hf":{"mode":"chat","base_model":"llama-3-70b-instruct","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v3-8b":{"mode":"chat","base_model":"llama-3-8b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v3-8b-instruct-hf":{"mode":"chat","base_model":"llama-3-8b-instruct","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v3p1-405b-instruct-long":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/llama-v3p1-405b-instruct-long","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/llama-v3p1-70b-instruct":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/llama-v3p1-70b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/llama-v3p1-70b-instruct-1b":{"mode":"chat","base_model":"llama-3.1-70b-instruct-1b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v3p1-nemotron-70b-instruct":{"mode":"chat","base_model":"llama-3.1-nemotron-70b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llama-v3p2-1b":{"mode":"chat","base_model":"llama-3.2-1b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v3p2-3b":{"mode":"chat","base_model":"llama-3.2-3b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/llama-v3p3-70b-instruct":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/llama-v3p3-70b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/llamaguard-7b":{"mode":"chat","base_model":"llamaguard-7b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/llava-yi-34b":{"mode":"chat","base_model":"llava-yi-34b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/minimax-m1-80k":{"mode":"chat","base_model":"minimax-m1-80k","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/minimax-m2":{"mode":"chat","base_model":"minimax-m2","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/ministral-3-14b-instruct-2512":{"mode":"chat","base_model":"ministral-3-14b-instruct","max_tokens":256000,"max_input_tokens":256000,"max_output_tokens":256000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/ministral-3-3b-instruct-2512":{"mode":"chat","base_model":"ministral-3-3b-instruct","max_tokens":256000,"max_input_tokens":256000,"max_output_tokens":256000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/ministral-3-8b-instruct-2512":{"mode":"chat","base_model":"ministral-3-8b-instruct","max_tokens":256000,"max_input_tokens":256000,"max_output_tokens":256000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mistral-7b":{"mode":"chat","base_model":"mistral-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-4k":{"mode":"chat","base_model":"mistral-7b-instruct-4k","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-v0p2":{"mode":"chat","base_model":"mistral-7b-instruct-v0.2","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mistral-7b-instruct-v3":{"mode":"chat","base_model":"mistral-7b-instruct-v3","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mistral-7b-v0p2":{"mode":"chat","base_model":"mistral-7b-v0.2","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mistral-large-3-fp8":{"mode":"chat","base_model":"mistral-large-3-fp8","max_tokens":256000,"max_input_tokens":256000,"max_output_tokens":256000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mistral-nemo-base-2407":{"mode":"chat","base_model":"mistral-nemo-base","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mistral-nemo-instruct-2407":{"mode":"chat","base_model":"mistral-nemo-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mistral-small-24b-instruct-2501":{"mode":"chat","base_model":"mistral-small-24b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mixtral-8x22b":{"mode":"chat","base_model":"mixtral-8x22b","max_tokens":65536,"max_input_tokens":65536,"max_output_tokens":65536,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":65536}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/mixtral-8x22b-instruct":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/mixtral-8x22b-instruct","max_tokens":64000,"max_input_tokens":64000,"max_output_tokens":64000,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/mixtral-8x7b":{"mode":"chat","base_model":"mixtral-8x7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/mixtral-8x7b-instruct":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mixtral-8x7b-instruct-hf":{"mode":"chat","base_model":"mixtral-8x7b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/mythomax-l2-13b":{"mode":"chat","base_model":"mythomax-l2-13b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nemotron-nano-v2-12b-vl":{"mode":"chat","base_model":"nemotron-nano-v2-12b-vl","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nous-capybara-7b-v1p9":{"mode":"chat","base_model":"nous-capybara-7b-v1.9","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nous-hermes-2-mixtral-8x7b-dpo":{"mode":"chat","base_model":"nous-hermes-2-mixtral-8x7b-dpo","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nous-hermes-2-yi-34b":{"mode":"chat","base_model":"nous-hermes-2-yi-34b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-13b":{"mode":"chat","base_model":"nous-hermes-llama2-13b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-70b":{"mode":"chat","base_model":"nous-hermes-llama2-70b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nous-hermes-llama2-7b":{"mode":"chat","base_model":"nous-hermes-llama2-7b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nvidia-nemotron-nano-12b-v2":{"mode":"chat","base_model":"nvidia-nemotron-nano-12b-v2","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/nvidia-nemotron-nano-9b-v2":{"mode":"chat","base_model":"nvidia-nemotron-nano-9b-v2","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/openchat-3p5-0106-7b":{"mode":"chat","base_model":"openchat-3.5-0106-7b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/openhermes-2-mistral-7b":{"mode":"chat","base_model":"openhermes-2-mistral-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/openhermes-2p5-mistral-7b":{"mode":"chat","base_model":"openhermes-2.5-mistral-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/openorca-7b":{"mode":"chat","base_model":"openorca-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/phi-2-3b":{"mode":"chat","base_model":"phi-2-3b","max_tokens":2048,"max_input_tokens":2048,"max_output_tokens":2048,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/phi-3-mini-128k-instruct":{"mode":"chat","base_model":"phi-3-mini-128k-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/phi-3-vision-128k-instruct":{"mode":"chat","base_model":"phi-3-vision-128k-instruct","max_tokens":32064,"max_input_tokens":32064,"max_output_tokens":32064,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32064}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-python-v1":{"mode":"chat","base_model":"phind-code-llama-34b-python-v1","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-v1":{"mode":"chat","base_model":"phind-code-llama-34b-v1","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/phind-code-llama-34b-v2":{"mode":"chat","base_model":"phind-code-llama-34b-v2","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/playground-v2-1024px-aesthetic":{"mode":"image_generation","base_model":"playground-v2-1024px-aesthetic","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/playground-v2-5-1024px-aesthetic":{"mode":"image_generation","base_model":"playground-v2-5-1024px-aesthetic","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/pythia-12b":{"mode":"chat","base_model":"pythia-12b","max_tokens":2048,"max_input_tokens":2048,"max_output_tokens":2048,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen-qwq-32b-preview":{"mode":"chat","base_model":"qwq-32b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen-v2p5-14b-instruct":{"mode":"chat","base_model":"qwen2.5-14b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen-v2p5-7b":{"mode":"chat","base_model":"qwen2.5-7b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen1p5-72b-chat":{"mode":"chat","base_model":"qwen1.5-72b-chat","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2-7b-instruct":{"mode":"chat","base_model":"qwen2-7b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2-vl-2b-instruct":{"mode":"chat","base_model":"qwen2-vl-2b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2-vl-72b-instruct":{"mode":"image_generation","base_model":"fireworks/accounts/fireworks/models/qwen2-vl-72b-instruct","max_tokens":32000,"max_input_tokens":32000,"max_output_tokens":32000,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"min_p","label":"Min P","helpText":"A number between 0 and 1 that can be used as an alternative to top_p and top-k","type":"number","default":0,"range":{"min":0,"max":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}],"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/qwen2-vl-7b-instruct":{"mode":"chat","base_model":"qwen2-vl-7b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-0p5b-instruct":{"mode":"chat","base_model":"qwen2.5-0.5b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-14b":{"mode":"chat","base_model":"qwen2.5-14b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-1p5b-instruct":{"mode":"chat","base_model":"qwen2.5-1.5b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-32b":{"mode":"chat","base_model":"qwen2.5-32b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-32b-instruct":{"mode":"chat","base_model":"qwen2.5-32b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-72b":{"mode":"chat","base_model":"qwen2.5-72b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-72b-instruct":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/qwen2p5-72b-instruct","max_tokens":32000,"max_input_tokens":32000,"max_output_tokens":32000,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/qwen2p5-7b-instruct":{"mode":"chat","base_model":"qwen2.5-7b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-0p5b":{"mode":"chat","base_model":"qwen2.5-coder-0.5b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-0p5b-instruct":{"mode":"chat","base_model":"qwen2.5-coder-0.5b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-14b":{"mode":"chat","base_model":"qwen2.5-coder-14b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-14b-instruct":{"mode":"chat","base_model":"qwen2.5-coder-14b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-1p5b":{"mode":"chat","base_model":"qwen2.5-coder-1.5b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-1p5b-instruct":{"mode":"chat","base_model":"qwen2.5-coder-1.5b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b":{"mode":"chat","base_model":"qwen2.5-coder-32b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-128k":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct-128k","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-32k-rope":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct-32k-rope","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-32b-instruct-64k":{"mode":"chat","base_model":"qwen2.5-coder-32b-instruct-64k","max_tokens":65536,"max_input_tokens":65536,"max_output_tokens":65536,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-3b":{"mode":"chat","base_model":"qwen2.5-coder-3b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-3b-instruct":{"mode":"chat","base_model":"qwen2.5-coder-3b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-7b":{"mode":"chat","base_model":"qwen2.5-coder-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-coder-7b-instruct":{"mode":"chat","base_model":"qwen2.5-coder-7b-instruct","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-math-72b-instruct":{"mode":"chat","base_model":"qwen2.5-math-72b-instruct","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-vl-32b-instruct":{"mode":"image_generation","base_model":"fireworks/accounts/fireworks/models/qwen2p5-vl-32b-instruct","max_tokens":125000,"max_input_tokens":125000,"max_output_tokens":125000,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":125000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"min_p","label":"Min P","helpText":"A number between 0 and 1 that can be used as an alternative to top_p and top-k","type":"number","default":0,"range":{"min":0,"max":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}],"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/qwen2p5-vl-3b-instruct":{"mode":"chat","base_model":"qwen2.5-vl-3b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-vl-72b-instruct":{"mode":"chat","base_model":"qwen2.5-vl-72b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen2p5-vl-7b-instruct":{"mode":"chat","base_model":"qwen2.5-vl-7b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-0p6b":{"mode":"chat","base_model":"qwen3-0.6b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":40960}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen3-14b":{"mode":"chat","base_model":"qwen3-14b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":40960}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen3-1p7b":{"mode":"chat","base_model":"qwen3-1.7b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft":{"mode":"chat","base_model":"qwen3-1.7b-fp8-draft","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft-131072":{"mode":"chat","base_model":"qwen3-1.7b-fp8-draft-131072","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-1p7b-fp8-draft-40960":{"mode":"chat","base_model":"qwen3-1.7b-fp8-draft-40960","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/qwen3-235b-a22b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-235b-a22b-thinking-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/qwen3-30b-a3b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-instruct-2507":{"mode":"chat","base_model":"qwen3-30b-a3b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-30b-a3b-thinking-2507":{"mode":"chat","base_model":"qwen3-30b-a3b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-32b":{"mode":"chat","base_model":"qwen3-32b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_reasoning":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":131072}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen3-4b":{"mode":"chat","base_model":"qwen3-4b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":40960}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen3-4b-instruct-2507":{"mode":"chat","base_model":"qwen3-4b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-8b":{"mode":"chat","base_model":"qwen3-8b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"supports_reasoning":true,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":40960}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen3-coder-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-coder-30b-a3b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-coder-480b-instruct-bf16":{"mode":"chat","base_model":"qwen3-coder-480b-instruct-bf16","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-embedding-0p6b":{"mode":"embedding","base_model":"qwen3-embedding-0.6b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai"},"fireworks_ai/accounts/fireworks/models/qwen3-embedding-4b":{"mode":"embedding","base_model":"qwen3-embedding-4b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"provider":"fireworks_ai"},"fireworks_ai/accounts/fireworks/models/":{"mode":"embedding","base_model":"","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"provider":"fireworks_ai"},"fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-next-80b-a3b-thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-reranker-0p6b":{"mode":"rerank","base_model":"qwen3-reranker-0.6b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-reranker-4b":{"mode":"rerank","base_model":"qwen3-reranker-4b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-reranker-8b":{"mode":"rerank","base_model":"qwen3-reranker-8b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"source":"https://api.fireworks.ai/v1/serverless/models","provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-vl-235b-a22b-instruct":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-vl-235b-a22b-thinking":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-vl-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-vl-30b-a3b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-vl-30b-a3b-thinking":{"mode":"chat","base_model":"qwen3-vl-30b-a3b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-vl-32b-instruct":{"mode":"chat","base_model":"qwen3-vl-32b-instruct","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwen3-vl-8b-instruct":{"mode":"chat","base_model":"qwen3-vl-8b-instruct","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/qwq-32b":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/qwq-32b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/rolm-ocr":{"mode":"chat","base_model":"rolm-ocr","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/snorkel-mistral-7b-pairrm-dpo":{"mode":"chat","base_model":"snorkel-mistral-7b-pairrm-dpo","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/stable-diffusion-xl-1024-v1-0":{"mode":"image_generation","base_model":"stable-diffusion-xl-1024-v1-0","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/stablecode-3b":{"mode":"chat","base_model":"stablecode-3b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/starcoder-16b":{"mode":"chat","base_model":"starcoder-16b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/starcoder-7b":{"mode":"chat","base_model":"starcoder-7b","max_tokens":8192,"max_input_tokens":8192,"max_output_tokens":8192,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/starcoder2-15b":{"mode":"chat","base_model":"starcoder2-15b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/starcoder2-3b":{"mode":"chat","base_model":"starcoder2-3b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/starcoder2-7b":{"mode":"chat","base_model":"starcoder2-7b","max_tokens":16384,"max_input_tokens":16384,"max_output_tokens":16384,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/toppy-m-7b":{"mode":"chat","base_model":"toppy-m-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/whisper-v3":{"mode":"audio_transcription","base_model":"whisper-v3","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/whisper-v3-turbo":{"mode":"audio_transcription","base_model":"whisper-v3-turbo","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/yi-34b":{"mode":"chat","base_model":"yi-34b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/yi-34b-200k-capybara":{"mode":"chat","base_model":"yi-34b-200k-capybara","max_tokens":200000,"max_input_tokens":200000,"max_output_tokens":200000,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"fireworks_ai/accounts/fireworks/models/yi-34b-chat":{"mode":"chat","base_model":"yi-34b-chat","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/yi-6b":{"mode":"chat","base_model":"yi-6b","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/zephyr-7b-beta":{"mode":"chat","base_model":"zephyr-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"provider":"fireworks_ai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/deepseek/deepseek-v3.2":{"mode":"chat","base_model":"deepseek","max_input_tokens":163840,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/minimax/minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/zai-org/glm-4.7":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/xiaomimimo/mimo-v2-flash":{"mode":"chat","base_model":"mimo-v2-flash","max_input_tokens":262144,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/zai-org/autoglm-phone-9b-multilingual":{"mode":"chat","base_model":"autoglm-phone-9b-multilingual","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"supports_vision":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/moonshotai/kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"supports_prompt_caching":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/minimax/minimax-m2":{"mode":"chat","base_model":"minimax-m2","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/paddlepaddle/paddleocr-vl":{"mode":"chat","base_model":"paddleocr-vl","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"supports_vision":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-v3.2-exp":{"mode":"chat","base_model":"deepseek-v3.2","max_input_tokens":163840,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-vl-235b-a22b-thinking":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_vision":true,"supports_system_messages":true,"supports_reasoning":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/zai-org/glm-4.6v":{"mode":"chat","base_model":"glm-4.6v","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/zai-org/glm-4.6":{"mode":"chat","base_model":"glm-4.6","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/kwaipilot/kat-coder-pro":{"mode":"chat","base_model":"kat-coder-pro","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-next-80b-a3b-instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-next-80b-a3b-thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-ocr":{"mode":"chat","base_model":"deepseek-ocr","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-v3.1-terminus":{"mode":"chat","base_model":"deepseek-v3.1-terminus","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-vl-235b-a22b-instruct":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-max":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/skywork/r1v4-lite":{"mode":"chat","base_model":"r1v4-lite","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-v3.1":{"mode":"chat","base_model":"deepseek","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/moonshotai/kimi-k2-0905":{"mode":"chat","base_model":"kimi-k2","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-coder-480b-a35b-instruct":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-coder-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-coder-30b-a3b-instruct","max_input_tokens":160000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/moonshotai/kimi-k2-instruct":{"mode":"chat","base_model":"kimi-k2-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-v3-0324":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/zai-org/glm-4.5":{"mode":"chat","base_model":"glm-4.5","max_input_tokens":131072,"max_output_tokens":98304,"max_tokens":98304,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":98304}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-235b-a22b-thinking-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/meta-llama/llama-3.1-8b-instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/google/gemma-3-12b-it":{"mode":"chat","base_model":"gemma-3-12b-it","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/zai-org/glm-4.5v":{"mode":"chat","base_model":"glm-4.5v","max_input_tokens":65536,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/qwen/qwen3-235b-a22b-instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-r1-distill-qwen-14b":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-14b","max_input_tokens":32768,"max_output_tokens":16384,"max_tokens":16384,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/meta-llama/llama-3.3-70b-instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_input_tokens":131072,"max_output_tokens":120000,"max_tokens":120000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":120000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen-2.5-72b-instruct":{"mode":"chat","base_model":"qwen2.5-72b-instruct","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/mistralai/mistral-nemo":{"mode":"chat","base_model":"mistral-nemo","max_input_tokens":60288,"max_output_tokens":16000,"max_tokens":16000,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/minimaxai/minimax-m1-80k":{"mode":"chat","base_model":"minimax-m1-80k","max_input_tokens":1000000,"max_output_tokens":40000,"max_tokens":40000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-r1-0528":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":163840,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-r1-distill-qwen-32b":{"mode":"chat","base_model":"deepseek-r1-distill-qwen-32b","max_input_tokens":64000,"max_output_tokens":32000,"max_tokens":32000,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/meta-llama/llama-3-8b-instruct":{"mode":"chat","base_model":"llama-3-8b-instruct","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/microsoft/wizardlm-2-8x22b":{"mode":"chat","base_model":"wizardlm-2-8x22b","max_input_tokens":65535,"max_output_tokens":8000,"max_tokens":8000,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/deepseek/deepseek-r1-0528-qwen3-8b":{"mode":"chat","base_model":"deepseek-r1-0528-qwen3-8b","max_input_tokens":128000,"max_output_tokens":32000,"max_tokens":32000,"supports_system_messages":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-r1-distill-llama-70b":{"mode":"chat","base_model":"deepseek-r1-distill-llama-70b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/meta-llama/llama-3-70b-instruct":{"mode":"chat","base_model":"llama-3-70b-instruct","max_input_tokens":8192,"max_output_tokens":8000,"max_tokens":8000,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/qwen/qwen3-235b-a22b-fp8":{"mode":"chat","base_model":"qwen3-235b-a22b-fp8","max_input_tokens":40960,"max_output_tokens":20000,"max_tokens":20000,"supports_system_messages":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":20000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct-fp8","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/meta-llama/llama-4-scout-17b-16e-instruct":{"mode":"chat","base_model":"llama-4-scout-17b-16e-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_vision":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/nousresearch/hermes-2-pro-llama-3-8b":{"mode":"chat","base_model":"hermes-2-pro-llama-3-8b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen2.5-vl-72b-instruct":{"mode":"chat","base_model":"qwen2.5-vl-72b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supports_vision":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/sao10k/l3-70b-euryale-v2.1":{"mode":"chat","base_model":"l3-70b-euryale","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/baidu/ernie-4.5-21B-a3b-thinking":{"mode":"chat","base_model":"ernie-4.5-21b-a3b-thinking","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"supports_system_messages":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/sao10k/l3-8b-lunaris":{"mode":"chat","base_model":"l3-8b-lunaris","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/baichuan/baichuan-m2-32b":{"mode":"chat","base_model":"baichuan-m2-32b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/baidu/ernie-4.5-vl-424b-a47b":{"mode":"chat","base_model":"ernie-4.5-vl-424b-a47b","max_input_tokens":123000,"max_output_tokens":16000,"max_tokens":16000,"supports_vision":true,"supports_system_messages":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/baidu/ernie-4.5-300b-a47b-paddle":{"mode":"chat","base_model":"ernie-4.5-300b-a47b-paddle","max_input_tokens":123000,"max_output_tokens":12000,"max_tokens":12000,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":12000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-prover-v2-671b":{"mode":"chat","base_model":"deepseek-prover-v2-671b","max_input_tokens":160000,"max_output_tokens":160000,"max_tokens":160000,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":160000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-32b-fp8":{"mode":"chat","base_model":"qwen3-32b-fp8","max_input_tokens":40960,"max_output_tokens":20000,"max_tokens":20000,"supports_system_messages":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":20000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-30b-a3b-fp8":{"mode":"chat","base_model":"qwen3-30b-a3b-fp8","max_input_tokens":40960,"max_output_tokens":20000,"max_tokens":20000,"supports_system_messages":true,"supports_reasoning":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":20000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/google/gemma-3-27b-it":{"mode":"chat","base_model":"gemma-3-27b-it","max_input_tokens":98304,"max_output_tokens":16384,"max_tokens":16384,"supports_vision":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-v3-turbo":{"mode":"chat","base_model":"deepseek-v3-turbo","max_input_tokens":64000,"max_output_tokens":16000,"max_tokens":16000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/deepseek/deepseek-r1-turbo":{"mode":"chat","base_model":"deepseek-r1-turbo","max_input_tokens":64000,"max_output_tokens":16000,"max_tokens":16000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/Sao10K/L3-8B-Stheno-v3.2":{"mode":"chat","base_model":"l3-8b-stheno","max_input_tokens":8192,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/gryphe/mythomax-l2-13b":{"mode":"chat","base_model":"mythomax-l2-13b","max_input_tokens":4096,"max_output_tokens":3200,"max_tokens":3200,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":3200}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/baidu/ernie-4.5-vl-28b-a3b-thinking":{"mode":"chat","base_model":"ernie-4.5-vl-28b-a3b-thinking","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-vl-8b-instruct":{"mode":"chat","base_model":"qwen3-vl-8b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/zai-org/glm-4.5-air":{"mode":"chat","base_model":"glm-4.5-air","max_input_tokens":131072,"max_output_tokens":98304,"max_tokens":98304,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_reasoning":true,"supports_prompt_caching":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":98304}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-vl-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-vl-30b-a3b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-vl-30b-a3b-thinking":{"mode":"chat","base_model":"qwen3-vl-30b-a3b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-omni-30b-a3b-thinking":{"mode":"chat","base_model":"qwen3-omni-30b-a3b-thinking","max_input_tokens":65536,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"supports_reasoning":true,"supports_audio_input":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-omni-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-omni-30b-a3b-instruct","max_input_tokens":65536,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_response_schema":true,"supports_audio_input":true,"supports_audio_output":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen-mt-plus":{"mode":"chat","base_model":"qwen-mt-plus","max_input_tokens":16384,"max_output_tokens":8192,"max_tokens":8192,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/baidu/ernie-4.5-vl-28b-a3b":{"mode":"chat","base_model":"ernie-4.5-vl-28b-a3b","max_input_tokens":30000,"max_output_tokens":8000,"max_tokens":8000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_system_messages":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/baidu/ernie-4.5-21B-a3b":{"mode":"chat","base_model":"ernie-4.5-21b-a3b","max_input_tokens":120000,"max_output_tokens":8000,"max_tokens":8000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-8b-fp8":{"mode":"chat","base_model":"qwen3-8b-fp8","max_input_tokens":128000,"max_output_tokens":20000,"max_tokens":20000,"supports_system_messages":true,"supports_reasoning":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":20000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-4b-fp8":{"mode":"chat","base_model":"qwen3-4b-fp8","max_input_tokens":128000,"max_output_tokens":20000,"max_tokens":20000,"supports_system_messages":true,"supports_reasoning":true,"supports_function_calling":true,"supports_tool_choice":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":20000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen2.5-7b-instruct":{"mode":"chat","base_model":"qwen2.5-7b-instruct","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"novita/meta-llama/llama-3.2-3b-instruct":{"mode":"chat","base_model":"llama-3.2-3b-instruct","max_input_tokens":32768,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/sao10k/l31-70b-euryale-v2.2":{"mode":"chat","base_model":"l31-70b-euryale","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/qwen/qwen3-embedding-0.6b":{"mode":"embedding","base_model":"qwen3-embedding-0.6b","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"provider":"novita"},"novita/qwen/qwen3-embedding-8b":{"mode":"embedding","base_model":"qwen3-embedding-8b","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"provider":"novita"},"novita/baai/bge-m3":{"mode":"embedding","base_model":"bge-m3","max_input_tokens":8192,"max_output_tokens":96000,"max_tokens":96000,"provider":"novita"},"novita/qwen/qwen3-reranker-8b":{"mode":"rerank","base_model":"qwen3-reranker-8b","max_input_tokens":32768,"max_output_tokens":4096,"max_tokens":4096,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"novita/baai/bge-reranker-v2-m3":{"mode":"rerank","base_model":"bge-reranker-v2-m3","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"provider":"novita","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"llamagate/llama-3.1-8b":{"mode":"chat","base_model":"llama-3.1-8b","max_tokens":8192,"max_input_tokens":131072,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/llama-3.2-3b":{"mode":"chat","base_model":"llama-3.2-3b","max_tokens":8192,"max_input_tokens":131072,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/mistral-7b-v0.3":{"mode":"chat","base_model":"mistral-7b","max_tokens":8192,"max_input_tokens":32768,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/qwen3-8b":{"mode":"chat","base_model":"qwen3-8b","max_tokens":8192,"max_input_tokens":32768,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/dolphin3-8b":{"mode":"chat","base_model":"dolphin3-8b","max_tokens":8192,"max_input_tokens":128000,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/deepseek-r1-8b":{"mode":"chat","base_model":"deepseek-r1-8b","max_tokens":16384,"max_input_tokens":65536,"max_output_tokens":16384,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/deepseek-r1-7b-qwen":{"mode":"chat","base_model":"deepseek-r1-7b-qwen","max_tokens":16384,"max_input_tokens":131072,"max_output_tokens":16384,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"llamagate/openthinker-7b":{"mode":"chat","base_model":"openthinker-7b","max_tokens":8192,"max_input_tokens":32768,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/qwen2.5-coder-7b":{"mode":"chat","base_model":"qwen2.5-coder-7b","max_tokens":8192,"max_input_tokens":32768,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/deepseek-coder-6.7b":{"mode":"chat","base_model":"deepseek-coder-6.7b","max_tokens":4096,"max_input_tokens":16384,"max_output_tokens":4096,"supports_function_calling":true,"supports_response_schema":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/codellama-7b":{"mode":"chat","base_model":"codellama-7b","max_tokens":4096,"max_input_tokens":16384,"max_output_tokens":4096,"supports_function_calling":true,"supports_response_schema":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":4096}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/qwen3-vl-8b":{"mode":"chat","base_model":"qwen3-vl-8b","max_tokens":8192,"max_input_tokens":32768,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"llamagate/llava-7b":{"mode":"chat","base_model":"llava-7b","max_tokens":2048,"max_input_tokens":4096,"max_output_tokens":2048,"supports_response_schema":true,"supports_vision":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":2048}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/gemma3-4b":{"mode":"chat","base_model":"gemma-3-4b","max_tokens":8192,"max_input_tokens":128000,"max_output_tokens":8192,"supports_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"llamagate","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"llamagate/nomic-embed-text":{"mode":"embedding","base_model":"nomic-embed-text","max_tokens":8192,"max_input_tokens":8192,"provider":"llamagate"},"llamagate/qwen3-embedding-8b":{"mode":"embedding","base_model":"qwen3-embedding-8b","max_tokens":40960,"max_input_tokens":40960,"provider":"llamagate"},"sarvam/sarvam-m":{"mode":"chat","base_model":"sarvam-m","max_input_tokens":8192,"max_output_tokens":32000,"max_tokens":32000,"supports_reasoning":true,"provider":"sarvam","is_deprecated":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"tts-1-1106":{"mode":"audio_speech","base_model":"tts-1","supported_endpoints":["/v1/audio/speech"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"tts-1-hd-1106":{"mode":"audio_speech","base_model":"tts-1-hd","supported_endpoints":["/v1/audio/speech"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-tts-2025-03-20":{"mode":"audio_speech","base_model":"gpt-4o-mini-tts","supported_endpoints":["/v1/audio/speech"],"supported_modalities":["text","audio"],"supported_output_modalities":["audio"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-tts-2025-12-15":{"mode":"audio_speech","base_model":"gpt-4o-mini-tts","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/audio/speech"],"supported_modalities":["text","audio"],"supported_output_modalities":["audio"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-transcribe-2025-03-20":{"mode":"audio_transcription","base_model":"gpt-4o-mini-transcribe","deprecation_date":"2027-01-20","max_input_tokens":16000,"max_output_tokens":2000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-4o-mini-transcribe-2025-12-15":{"mode":"audio_transcription","base_model":"gpt-4o-mini-transcribe","max_input_tokens":16000,"max_output_tokens":2000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-5-search-api":{"mode":"chat","base_model":"gpt-5-search-api","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":false,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_minimal_reasoning_effort":true,"provider":"openai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-5-search-api-2025-10-14":{"mode":"chat","base_model":"gpt-5-search-api","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supports_function_calling":false,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"provider":"openai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_reasoning":false,"supports_service_tier":true,"supports_assistant_prefill":true},"gpt-realtime-mini-2025-10-06":{"mode":"chat","base_model":"gpt-realtime-mini","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gpt-realtime-mini-2025-12-15":{"mode":"chat","base_model":"gpt-realtime-mini","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"deprecation_date":"2027-01-20","source":"https://developers.openai.com/api/docs/pricing","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sora-2":{"mode":"video_generation","base_model":"sora-2","deprecation_date":"2026-09-24","source":"https://platform.openai.com/docs/api-reference/videos","supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"openai","supported_resolutions":["720x1280","1280x720"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sora-2-pro":{"mode":"video_generation","base_model":"sora-2-pro","deprecation_date":"2026-09-24","source":"https://platform.openai.com/docs/api-reference/videos","supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"openai","supported_resolutions":["720x1280","1280x720"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"sora-2-pro-high-res":{"mode":"video_generation","base_model":"sora-2-pro","source":"https://platform.openai.com/docs/api-reference/videos","supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"openai","supported_resolutions":["1024x1792","1792x1024"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"chatgpt-image-latest":{"mode":"image_generation","base_model":"chatgpt-image","deprecation_date":"2026-12-01","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.0-flash-exp-image-generation":{"mode":"image_generation","base_model":"gemini-2.0-flash-exp-image-generation","max_images_per_prompt":3000,"max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://ai.google.dev/pricing","supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_vision":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.0-flash-exp-image-generation":{"mode":"image_generation","base_model":"gemini-2.0-flash-exp-image-generation","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://ai.google.dev/pricing","supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_vision":true,"tpm":250000,"rpm":10,"provider":"gemini","max_images_per_prompt":3000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.0-flash-lite-001":{"mode":"chat","base_model":"gemini-2.0-flash-lite","deprecation_date":"2026-03-31","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":8192,"max_pdf_size_mb":50,"max_video_length":1,"max_videos_per_prompt":10,"rpm":4000,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-2.0-flash-lite","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tpm":4000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"is_deprecated":true},"gemini-2.5-flash-native-audio-latest":{"mode":"chat","base_model":"gemini-2.5-flash-native-audio","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-flash-native-audio-preview-09-2025":{"mode":"chat","base_model":"gemini-2.5-flash-native-audio","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-flash-native-audio-preview-12-2025":{"mode":"chat","base_model":"gemini-2.5-flash-native-audio","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.5-flash-native-audio-latest":{"mode":"chat","base_model":"gemini-2.5-flash-native-audio","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"tpm":250000,"rpm":10,"gemini_native_audio":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.5-flash-native-audio-preview-09-2025":{"mode":"chat","base_model":"gemini-2.5-flash-native-audio","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"tpm":250000,"rpm":10,"gemini_native_audio":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini/gemini-2.5-flash-native-audio-preview-12-2025":{"mode":"chat","base_model":"gemini-2.5-flash-native-audio","max_input_tokens":1048576,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"tpm":250000,"rpm":10,"gemini_native_audio":true,"supports_function_calling":true,"supports_response_schema":false,"supports_vision":true,"supports_web_search":true,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-2.5-flash-preview-tts":{"mode":"audio_speech","base_model":"gemini-2.5-flash-tts","source":"https://ai.google.dev/pricing","supported_endpoints":["/v1/audio/speech"],"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-flash-latest":{"mode":"chat","base_model":"gemini-flash","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":100000,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":8000000,"provider":"gemini","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":false,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high"],"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini-flash-lite-latest":{"mode":"chat","base_model":"gemini-flash-lite","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":15,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-lite","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"provider":"gemini","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":false,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["minimal","low","medium","high"],"supports_reasoning_disable":false,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini-pro-latest":{"mode":"chat","base_model":"gemini-pro","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":2000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"tpm":800000,"provider":"gemini","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high"],"supports_reasoning_disable":false,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini/gemini-pro-latest":{"mode":"chat","base_model":"gemini-pro","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":2000,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"tpm":800000,"prompt_cache_min_tokens":4096,"supports_native_streaming":true,"supports_url_context":true,"web_search_billing_unit":"per_query","provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-exp-1206":{"mode":"chat","base_model":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_input_tokens":1048576,"max_output_tokens":65535,"max_pdf_size_mb":30,"max_tokens":65535,"max_video_length":1,"max_videos_per_prompt":10,"rpm":100000,"source":"https://ai.google.dev/gemini-api/docs/models#gemini-2.5-flash-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_vision":true,"supports_web_search":true,"tpm":8000000,"provider":"gemini","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"cerebras/qwen-3-235b-a22b-instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"provider":"cerebras","supports_function_calling":true,"supports_vision":false,"supports_reasoning":false,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"fireworks_ai/accounts/fireworks/models/qwen3-embedding-8b":{"mode":"embedding","base_model":"qwen3-embedding-8b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":0,"provider":"fireworks_ai"},"zero-one-ai/Yi-34B-Chat":{"mode":"chat","base_model":"zero-one-ai/Yi-34B-Chat","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","accesorKey":"type","default":{"type":"medium"},"options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"source":"merged_from_llm_models_csv"},"Austism/chronos-hermes-13b":{"mode":"chat","base_model":"Austism/chronos-hermes-13b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"DiscoResearch/DiscoLM-mixtral-8x7b-v2":{"mode":"chat","base_model":"DiscoResearch/DiscoLM-mixtral-8x7b-v2","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Gryphe/MythoMax-L2-13b":{"mode":"chat","base_model":"Gryphe/MythoMax-L2-13b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"lmsys/vicuna-13b-v1.5":{"mode":"chat","base_model":"lmsys/vicuna-13b-v1.5","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"lmsys/vicuna-7b-v1.5":{"mode":"chat","base_model":"lmsys/vicuna-7b-v1.5","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"lmsys/vicuna-13b-v1.5-16k":{"mode":"chat","base_model":"lmsys/vicuna-13b-v1.5-16k","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-13b-Instruct-hf":{"mode":"chat","base_model":"codellama/CodeLlama-13b-Instruct-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-34b-Instruct-hf":{"mode":"chat","base_model":"codellama/CodeLlama-34b-Instruct-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-70b-Instruct-hf":{"mode":"chat","base_model":"codellama/CodeLlama-70b-Instruct-hf","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-7b-Instruct-hf":{"mode":"chat","base_model":"codellama/CodeLlama-7b-Instruct-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/llama-2-13b-chat":{"mode":"chat","base_model":"togethercomputer/llama-2-13b-chat","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/llama-2-70b-chat":{"mode":"chat","base_model":"togethercomputer/llama-2-70b-chat","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/llama-2-7b-chat":{"mode":"chat","base_model":"togethercomputer/llama-2-7b-chat","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Capybara-7B-V1p9":{"mode":"chat","base_model":"NousResearch/Nous-Capybara-7B-V1p9","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Hermes-2-Mixtral-8x7B-DPO":{"mode":"chat","base_model":"NousResearch/Nous-Hermes-2-Mixtral-8x7B-DPO","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Hermes-2-Mixtral-8x7B-SFT":{"mode":"chat","base_model":"NousResearch/Nous-Hermes-2-Mixtral-8x7B-SFT","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Hermes-Llama2-70b":{"mode":"chat","base_model":"NousResearch/Nous-Hermes-Llama2-70b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Hermes-llama-2-7b":{"mode":"chat","base_model":"NousResearch/Nous-Hermes-llama-2-7b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Hermes-Llama2-13b":{"mode":"chat","base_model":"NousResearch/Nous-Hermes-Llama2-13b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Hermes-2-Yi-34B":{"mode":"chat","base_model":"NousResearch/Nous-Hermes-2-Yi-34B","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"openchat/openchat-3.5-1210":{"mode":"chat","base_model":"openchat/openchat-3.5-1210","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Open-Orca/Mistral-7B-OpenOrca":{"mode":"chat","base_model":"Open-Orca/Mistral-7B-OpenOrca","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/Qwen-7B-Chat":{"mode":"chat","base_model":"togethercomputer/Qwen-7B-Chat","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"snorkelai/Snorkel-Mistral-PairRM-DPO":{"mode":"chat","base_model":"snorkelai/Snorkel-Mistral-PairRM-DPO","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/alpaca-7b":{"mode":"chat","base_model":"togethercomputer/alpaca-7b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/falcon-40b-instruct":{"mode":"chat","base_model":"togethercomputer/falcon-40b-instruct","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/falcon-7b-instruct":{"mode":"chat","base_model":"togethercomputer/falcon-7b-instruct","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/GPT-NeoXT-Chat-Base-20B":{"mode":"chat","base_model":"togethercomputer/GPT-NeoXT-Chat-Base-20B","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/Llama-2-7B-32K-Instruct":{"mode":"chat","base_model":"togethercomputer/Llama-2-7B-32K-Instruct","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/Pythia-Chat-Base-7B-v0.16":{"mode":"chat","base_model":"togethercomputer/Pythia-Chat-Base-7B-v0.16","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/RedPajama-INCITE-Chat-3B-v1":{"mode":"chat","base_model":"togethercomputer/RedPajama-INCITE-Chat-3B-v1","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/RedPajama-INCITE-7B-Chat":{"mode":"chat","base_model":"togethercomputer/RedPajama-INCITE-7B-Chat","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/StripedHyena-Nous-7B":{"mode":"chat","base_model":"togethercomputer/StripedHyena-Nous-7B","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Undi95/ReMM-SLERP-L2-13B":{"mode":"chat","base_model":"Undi95/ReMM-SLERP-L2-13B","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"Undi95/Toppy-M-7B":{"mode":"chat","base_model":"Undi95/Toppy-M-7B","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"WizardLM/WizardLM-13B-V1.2":{"mode":"chat","base_model":"WizardLM/WizardLM-13B-V1.2","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"garage-bAInd/Platypus2-70B-instruct":{"mode":"chat","base_model":"garage-bAInd/Platypus2-70B-instruct","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"mistralai/Mistral-7B-Instruct-v0.2":{"mode":"chat","base_model":"mistralai/Mistral-7B-Instruct-v0.2","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"teknium/OpenHermes-2-Mistral-7B":{"mode":"chat","base_model":"teknium/OpenHermes-2-Mistral-7B","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"teknium/OpenHermes-2p5-Mistral-7B":{"mode":"chat","base_model":"teknium/OpenHermes-2p5-Mistral-7B","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"upstage/SOLAR-10.7B-Instruct-v1.0":{"mode":"chat","base_model":"upstage/SOLAR-10.7B-Instruct-v1.0","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"zero-one-ai/Yi-34B":{"mode":"chat","base_model":"zero-one-ai/Yi-34B","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"zero-one-ai/Yi-6B":{"mode":"chat","base_model":"zero-one-ai/Yi-6B","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"EleutherAI/llemma_7b":{"mode":"chat","base_model":"EleutherAI/llemma_7b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"huggyllama/llama-65b":{"mode":"chat","base_model":"huggyllama/llama-65b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/llama-2-13b":{"mode":"chat","base_model":"togethercomputer/llama-2-13b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/llama-2-70b":{"mode":"chat","base_model":"togethercomputer/llama-2-70b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/llama-2-7b":{"mode":"chat","base_model":"togethercomputer/llama-2-7b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"microsoft/phi-2":{"mode":"chat","base_model":"microsoft/phi-2","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Nexusflow/NexusRaven-V2-13B":{"mode":"chat","base_model":"Nexusflow/NexusRaven-V2-13B","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/Qwen-7B":{"mode":"chat","base_model":"togethercomputer/Qwen-7B","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/falcon-40b":{"mode":"chat","base_model":"togethercomputer/falcon-40b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/falcon-7b":{"mode":"chat","base_model":"togethercomputer/falcon-7b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/GPT-JT-6B-v1":{"mode":"chat","base_model":"togethercomputer/GPT-JT-6B-v1","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/GPT-JT-Moderation-6B":{"mode":"chat","base_model":"togethercomputer/GPT-JT-Moderation-6B","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/LLaMA-2-7B-32K":{"mode":"chat","base_model":"togethercomputer/LLaMA-2-7B-32K","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/RedPajama-INCITE-Base-3B-v1":{"mode":"chat","base_model":"togethercomputer/RedPajama-INCITE-Base-3B-v1","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/RedPajama-INCITE-7B-Base":{"mode":"chat","base_model":"togethercomputer/RedPajama-INCITE-7B-Base","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/RedPajama-INCITE-Instruct-3B-v1":{"mode":"chat","base_model":"togethercomputer/RedPajama-INCITE-Instruct-3B-v1","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/RedPajama-INCITE-7B-Instruct":{"mode":"chat","base_model":"togethercomputer/RedPajama-INCITE-7B-Instruct","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/StripedHyena-Hessian-7B":{"mode":"chat","base_model":"togethercomputer/StripedHyena-Hessian-7B","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"WizardLM/WizardLM-70B-V1.0":{"mode":"chat","base_model":"WizardLM/WizardLM-70B-V1.0","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"mistralai/Mistral-7B-v0.1":{"mode":"chat","base_model":"mistralai/Mistral-7B-v0.1","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"mistralai/Mixtral-8x7B-v0.1":{"mode":"chat","base_model":"mistralai/Mixtral-8x7B-v0.1","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-13b-hf":{"mode":"chat","base_model":"codellama/CodeLlama-13b-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-34b-hf":{"mode":"chat","base_model":"codellama/CodeLlama-34b-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-70b-hf":{"mode":"chat","base_model":"codellama/CodeLlama-70b-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-7b-hf":{"mode":"chat","base_model":"codellama/CodeLlama-7b-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-13b-Python-hf":{"mode":"chat","base_model":"codellama/CodeLlama-13b-Python-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-34b-Python-hf":{"mode":"chat","base_model":"codellama/CodeLlama-34b-Python-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-70b-Python-hf":{"mode":"chat","base_model":"codellama/CodeLlama-70b-Python-hf","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"codellama/CodeLlama-7b-Python-hf":{"mode":"chat","base_model":"codellama/CodeLlama-7b-Python-hf","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NumbersStation/nsql-llama-2-7B":{"mode":"chat","base_model":"NumbersStation/nsql-llama-2-7B","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Phind/Phind-CodeLlama-34B-Python-v1":{"mode":"chat","base_model":"Phind/Phind-CodeLlama-34B-Python-v1","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Phind/Phind-CodeLlama-34B-v2":{"mode":"chat","base_model":"Phind/Phind-CodeLlama-34B-v2","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"WizardLM/WizardCoder-Python-34B-V1.0":{"mode":"chat","base_model":"WizardLM/WizardCoder-Python-34B-V1.0","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"WizardLM/WizardCoder-15B-V1.0":{"mode":"chat","base_model":"WizardLM/WizardCoder-15B-V1.0","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"prompthero/openjourney":{"mode":"image_generation","base_model":"prompthero/openjourney","provider":"together_ai","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"runwayml/stable-diffusion-v1-5":{"mode":"image_generation","base_model":"runwayml/stable-diffusion-v1-5","provider":"together_ai","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"SG161222/Realistic_Vision_V3.0_VAE":{"mode":"image_generation","base_model":"SG161222/Realistic_Vision_V3.0_VAE","provider":"together_ai","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"stabilityai/stable-diffusion-2-1":{"mode":"image_generation","base_model":"stabilityai/stable-diffusion-2-1","provider":"together_ai","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"stabilityai/stable-diffusion-xl-base-1.0":{"mode":"image_generation","base_model":"stabilityai/stable-diffusion-xl-base-1.0","provider":"together_ai","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"wavymulder/Analog-Diffusion":{"mode":"image_generation","base_model":"wavymulder/Analog-Diffusion","provider":"together_ai","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Meta-Llama/Llama-Guard-7b":{"mode":"moderation","base_model":"Meta-Llama/Llama-Guard-7b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"databricks/dolly-v2-12b":{"mode":"chat","base_model":"databricks/dolly-v2-12b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"databricks/dolly-v2-3b":{"mode":"chat","base_model":"databricks/dolly-v2-3b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"databricks/dolly-v2-7b":{"mode":"chat","base_model":"databricks/dolly-v2-7b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"HuggingFaceH4/zephyr-7b-beta":{"mode":"chat","base_model":"HuggingFaceH4/zephyr-7b-beta","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"HuggingFaceH4/starchat-alpha":{"mode":"chat","base_model":"HuggingFaceH4/starchat-alpha","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"OpenAssistant/oasst-sft-4-pythia-12b-epoch-3.5":{"mode":"chat","base_model":"OpenAssistant/oasst-sft-4-pythia-12b-epoch-3.5","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"OpenAssistant/stablelm-7b-sft-v7-epoch-3":{"mode":"chat","base_model":"OpenAssistant/stablelm-7b-sft-v7-epoch-3","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/Koala-13B":{"mode":"chat","base_model":"togethercomputer/Koala-13B","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/Koala-7B":{"mode":"chat","base_model":"togethercomputer/Koala-7B","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"lmsys/vicuna-13b-v1.3":{"mode":"chat","base_model":"lmsys/vicuna-13b-v1.3","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"lmsys/vicuna-7b-v1.3":{"mode":"chat","base_model":"lmsys/vicuna-7b-v1.3","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"lmsys/fastchat-t5-3b-v1.0":{"mode":"chat","base_model":"lmsys/fastchat-t5-3b-v1.0","provider":"together_ai","max_input_tokens":512,"max_output_tokens":512,"max_tokens":512,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/mpt-30b-chat":{"mode":"chat","base_model":"togethercomputer/mpt-30b-chat","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/mpt-7b-chat":{"mode":"chat","base_model":"togethercomputer/mpt-7b-chat","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/guanaco-13b":{"mode":"chat","base_model":"togethercomputer/guanaco-13b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/guanaco-33b":{"mode":"chat","base_model":"togethercomputer/guanaco-33b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/guanaco-65b":{"mode":"chat","base_model":"togethercomputer/guanaco-65b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"togethercomputer/guanaco-7b":{"mode":"chat","base_model":"togethercomputer/guanaco-7b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"defog/sqlcoder":{"mode":"chat","base_model":"defog/sqlcoder","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"EleutherAI/gpt-j-6b":{"mode":"chat","base_model":"EleutherAI/gpt-j-6b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"EleutherAI/gpt-neox-20b":{"mode":"chat","base_model":"EleutherAI/gpt-neox-20b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"EleutherAI/pythia-12b-v0":{"mode":"chat","base_model":"EleutherAI/pythia-12b-v0","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"EleutherAI/pythia-1b-v0":{"mode":"chat","base_model":"EleutherAI/pythia-1b-v0","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"EleutherAI/pythia-2.8b-v0":{"mode":"chat","base_model":"EleutherAI/pythia-2.8b-v0","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"EleutherAI/pythia-6.9b":{"mode":"chat","base_model":"EleutherAI/pythia-6.9b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"google/flan-t5-xl":{"mode":"chat","base_model":"google/flan-t5-xl","provider":"together_ai","max_input_tokens":512,"max_output_tokens":512,"max_tokens":512,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"google/flan-t5-xxl":{"mode":"chat","base_model":"google/flan-t5-xxl","provider":"together_ai","max_input_tokens":512,"max_output_tokens":512,"max_tokens":512,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"huggyllama/llama-13b":{"mode":"chat","base_model":"huggyllama/llama-13b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"huggyllama/llama-30b":{"mode":"chat","base_model":"huggyllama/llama-30b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"huggyllama/llama-7b":{"mode":"chat","base_model":"huggyllama/llama-7b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"mosaicml/mpt-7b":{"mode":"chat","base_model":"mosaicml/mpt-7b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"mosaicml/mpt-7b-instruct":{"mode":"chat","base_model":"mosaicml/mpt-7b-instruct","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Hermes-13b":{"mode":"chat","base_model":"NousResearch/Nous-Hermes-13b","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NumbersStation/nsql-6B":{"mode":"chat","base_model":"NumbersStation/nsql-6B","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"stabilityai/stablelm-base-alpha-3b":{"mode":"chat","base_model":"stabilityai/stablelm-base-alpha-3b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"stabilityai/stablelm-base-alpha-7b":{"mode":"chat","base_model":"stabilityai/stablelm-base-alpha-7b","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"bigcode/starcoder":{"mode":"chat","base_model":"bigcode/starcoder","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Replit-Code-v1 (3B)":{"mode":"chat","base_model":"Replit-Code-v1 (3B)","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Salesforce/codegen2-16B":{"mode":"chat","base_model":"Salesforce/codegen2-16B","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"Salesforce/codegen2-7B":{"mode":"chat","base_model":"Salesforce/codegen2-7B","provider":"together_ai","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"gpt-4-turbo-vision-128k":{"mode":"image_generation","base_model":"gpt-4-turbo-vision-128k","provider":"azure","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"DeepSeek-R1-0528":{"mode":"chat","base_model":"DeepSeek-R1-0528","provider":"azure","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"DeepSeek-V3-0324":{"mode":"chat","base_model":"DeepSeek-V3-0324","provider":"azure","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/anthropic/models/claude-3-5-haiku":{"mode":"image_generation","base_model":"publishers/anthropic/models/claude-3-5-haiku","provider":"vertex_ai","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"anthropic_version","label":"Anthropic Version","helpText":"The version of the Anthropic API to use. default is vertex-2023-10-16","type":"text","default":"vertex-2023-10-16"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/anthropic/models/claude-3-7-sonnet":{"mode":"image_generation","base_model":"publishers/anthropic/models/claude-3-7-sonnet","provider":"vertex_ai","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"anthropic_version","label":"Anthropic Version","helpText":"The version of the Anthropic API to use. default is vertex-2023-10-16","type":"text","default":"vertex-2023-10-16"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/anthropic/models/claude-opus-4":{"mode":"image_generation","base_model":"publishers/anthropic/models/claude-opus-4","provider":"vertex_ai","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"anthropic_version","label":"Anthropic Version","helpText":"The version of the Anthropic API to use. default is vertex-2023-10-16","type":"text","default":"vertex-2023-10-16"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/anthropic/models/claude-sonnet-4":{"mode":"image_generation","base_model":"publishers/anthropic/models/claude-sonnet-4","provider":"vertex_ai","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"anthropic_version","label":"Anthropic Version","helpText":"The version of the Anthropic API to use. default is vertex-2023-10-16","type":"text","default":"vertex-2023-10-16"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/anthropic/models/claude-opus-4-1":{"mode":"image_generation","base_model":"publishers/anthropic/models/claude-opus-4-1","provider":"vertex_ai","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"anthropic_version","label":"Anthropic Version","helpText":"The version of the Anthropic API to use. default is vertex-2023-10-16","type":"text","default":"vertex-2023-10-16"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/anthropic/models/claude-sonnet-4-5":{"mode":"image_generation","base_model":"publishers/anthropic/models/claude-sonnet-4-5","provider":"vertex_ai","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"anthropic_version","label":"Anthropic Version","helpText":"The version of the Anthropic API to use. default is vertex-2023-10-16","type":"text","default":"vertex-2023-10-16"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/anthropic/models/claude-haiku-4-5":{"mode":"image_generation","base_model":"publishers/anthropic/models/claude-haiku-4-5","provider":"vertex_ai","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1},"disabledCondition":{"paramId":"thinking","operator":"eq","value":true,"setValue":1,"disabledText":"Temperature is disabled when extended thinking is enabled."}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"label":"High","value":"high"},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"thinking","label":"Extended thinking","helpText":"Enable the extended thinking parameter","type":"boolean","default":true},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"anthropic_version","label":"Anthropic Version","helpText":"The version of the Anthropic API to use. default is vertex-2023-10-16","type":"text","default":"vertex-2023-10-16"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/anthropic/models/claude-opus-4-5":{"mode":"image_generation","base_model":"publishers/anthropic/models/claude-opus-4-5","provider":"vertex_ai","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1},"disabledCondition":{"paramId":"thinking","operator":"eq","value":true,"setValue":1,"disabledText":"Temperature is disabled when extended thinking is enabled."}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"thinking","label":"Extended thinking","helpText":"Enable the extended thinking parameter","type":"boolean","default":true},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"anthropic_version","label":"Anthropic Version","helpText":"The version of the Anthropic API to use. default is vertex-2023-10-16","type":"text","default":"vertex-2023-10-16"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.5-pro":{"mode":"image_generation","base_model":"publishers/google/models/gemini-2.5-pro","provider":"vertex_ai","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.0-flash-001":{"mode":"chat","base_model":"publishers/google/models/gemini-2.0-flash-001","provider":"vertex_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.0-flash-lite-001":{"mode":"chat","base_model":"publishers/google/models/gemini-2.0-flash-lite-001","provider":"vertex_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.5-flash-lite":{"mode":"chat","base_model":"publishers/google/models/gemini-2.5-flash-lite","provider":"vertex_ai","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.5-flash":{"mode":"chat","base_model":"publishers/google/models/gemini-2.5-flash","provider":"vertex_ai","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-3-pro-preview":{"mode":"chat","base_model":"publishers/google/models/gemini-3-pro-preview","provider":"vertex_ai","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-3-flash-preview":{"mode":"chat","base_model":"publishers/google/models/gemini-3-flash-preview","provider":"vertex_ai","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.5-flash-image":{"mode":"chat","base_model":"publishers/google/models/gemini-2.5-flash-image","provider":"vertex_ai","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-3-pro-image-preview":{"mode":"chat","base_model":"publishers/google/models/gemini-3-pro-image-preview","provider":"vertex_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.5-flash-lite-preview-09-2025":{"mode":"image_generation","base_model":"publishers/google/models/gemini-2.5-flash-lite-preview-09-2025","provider":"vertex_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.5-flash-preview-09-2025":{"mode":"image_generation","base_model":"publishers/google/models/gemini-2.5-flash-preview-09-2025","provider":"vertex_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"publishers/google/models/gemini-2.5-flash-image-preview":{"mode":"image_generation","base_model":"publishers/google/models/gemini-2.5-flash-image-preview","provider":"vertex_ai","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call. Supports Google Search, File Search, Code Execution, URL Context, and Function Calling.","type":"select"},{"id":"media_resolution","label":"Media Resolution","helpText":"Higher resolutions may provide better understanding but use more tokens.","type":"select","default":"media_resolution_medium","options":[{"label":"Low","value":"media_resolution_low"},{"label":"Medium","value":"media_resolution_medium"},{"label":"High","value":"media_resolution_high"}]},{"id":"temperature","label":"Temperature","helpText":"For Gemini 3, best results at default 1.0. Lower values may impact reasoning.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Output length","helpText":"Maximum number of tokens in response","type":"number","default":32768,"range":{"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":5,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-oss-120b-maas":{"mode":"chat","base_model":"openai/gpt-oss-120b-maas","provider":"vertex_ai","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_output_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":131072}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-oss-20b-maas":{"mode":"chat","base_model":"openai/gpt-oss-20b-maas","provider":"vertex_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_output_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"llama3-8b-8192":{"mode":"chat","base_model":"llama3-8b-8192","provider":"groq","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-70b-8192":{"mode":"chat","base_model":"llama3-70b-8192","provider":"groq","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-3.1-70b-versatile":{"mode":"chat","base_model":"llama-3.1-70b-versatile","provider":"groq","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-3.1-405b-reasoning":{"mode":"chat","base_model":"llama-3.1-405b-reasoning","provider":"groq","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"Qwen/Qwen1.5-72B-Chat":{"mode":"chat","base_model":"Qwen/Qwen1.5-72B-Chat","provider":"together_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"NousResearch/Nous-Hermes-2-Mistral-7B-DPO":{"mode":"chat","base_model":"NousResearch/Nous-Hermes-2-Mistral-7B-DPO","provider":"together_ai","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"deepseek-ai/deepseek-coder-33b-instruct":{"mode":"chat","base_model":"deepseek-ai/deepseek-coder-33b-instruct","provider":"together_ai","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"google/gemma-7b-it":{"mode":"chat","base_model":"google/gemma-7b-it","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"meta-llama/Llama-3.2-11B-Vision-Instruct-Turbo":{"mode":"chat","base_model":"meta-llama/Llama-3.2-11B-Vision-Instruct-Turbo","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"meta-llama/Llama-3-8b-chat-hf":{"mode":"chat","base_model":"meta-llama/Llama-3-8b-chat-hf","provider":"together_ai","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"meta-llama/Llama-3.2-90B-Vision-Instruct-Turbo":{"mode":"image_generation","base_model":"meta-llama/Llama-3.2-90B-Vision-Instruct-Turbo","provider":"together_ai","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"command-light-nightly":{"mode":"chat","base_model":"command-light-nightly","provider":"cohere","max_input_tokens":4000,"max_output_tokens":4000,"max_tokens":4000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4000}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":4000,"step":1}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek.r1-v1:0":{"mode":"chat","base_model":"deepseek.r1-v1:0","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-llm-r1-distill-qwen-7b":{"mode":"chat","base_model":"deepseek-llm-r1-distill-qwen-7b","provider":"bedrock","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-llm-r1-distill-qwen-32b":{"mode":"chat","base_model":"deepseek-llm-r1-distill-qwen-32b","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-llm-r1-distill-qwen-14b":{"mode":"chat","base_model":"deepseek-llm-r1-distill-qwen-14b","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-llm-r1-distill-llama-8b":{"mode":"chat","base_model":"deepseek-llm-r1-distill-llama-8b","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-llm-r1-distill-llama-70b":{"mode":"chat","base_model":"deepseek-llm-r1-distill-llama-70b","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"huggingface-llm-qwen2-5-coder-7b-instruct":{"mode":"chat","base_model":"huggingface-llm-qwen2-5-coder-7b-instruct","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"huggingface-llm-qwen2-5-coder-32b-instruct":{"mode":"chat","base_model":"huggingface-llm-qwen2-5-coder-32b-instruct","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"huggingface-llm-qwen2-5-7b-instruct":{"mode":"chat","base_model":"huggingface-llm-qwen2-5-7b-instruct","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"huggingface-llm-qwen2-5-72b-instruct":{"mode":"chat","base_model":"huggingface-llm-qwen2-5-72b-instruct","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"huggingface-llm-qwen2-5-32b-instruct":{"mode":"chat","base_model":"huggingface-llm-qwen2-5-32b-instruct","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"huggingface-llm-qwen2-5-14b-instruct":{"mode":"chat","base_model":"huggingface-llm-qwen2-5-14b-instruct","provider":"bedrock","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"ministral-8b-latest":{"mode":"chat","base_model":"ministral-8b-latest","provider":"mistral","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131000,"range":{"min":1,"max":131000}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-saba-latest":{"mode":"chat","base_model":"mistral-saba-latest","provider":"mistral","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32000,"range":{"min":1,"max":32000}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"ministral-3b-latest":{"mode":"chat","base_model":"ministral-3b-latest","provider":"mistral","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131000,"range":{"min":1,"max":131000}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"accounts/yi-01-ai/models/yi-large":{"mode":"chat","base_model":"accounts/yi-01-ai/models/yi-large","provider":"fireworks_ai","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"accounts/sentientfoundation/models/dobby-unhinged-llama-3-3-70b-new":{"mode":"chat","base_model":"accounts/sentientfoundation/models/dobby-unhinged-llama-3-3-70b-new","provider":"fireworks_ai","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"accounts/fireworks/models/alpha":{"mode":"chat","base_model":"accounts/fireworks/models/alpha","provider":"fireworks_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"accounts/fireworks/models/moa":{"mode":"chat","base_model":"accounts/fireworks/models/moa","provider":"fireworks_ai","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/horizon-beta":{"mode":"image_generation","base_model":"openrouter/horizon-beta","provider":"openrouter","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/codestral-2508":{"mode":"chat","base_model":"mistralai/codestral-2508","provider":"openrouter","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-30b-a3b-instruct-2507":{"mode":"chat","base_model":"qwen/qwen3-30b-a3b-instruct-2507","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"z-ai/glm-4.5-air:free":{"mode":"chat","base_model":"z-ai/glm-4.5-air:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"z-ai/glm-4-32b":{"mode":"chat","base_model":"z-ai/glm-4-32b","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-flash-lite":{"mode":"image_generation","base_model":"google/gemini-2.5-flash-lite","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":1048576}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"moonshotai/kimi-k2:free":{"mode":"chat","base_model":"moonshotai/kimi-k2:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moonshotai/kimi-k2":{"mode":"chat","base_model":"moonshotai/kimi-k2","provider":"openrouter","max_input_tokens":63000,"max_output_tokens":63000,"max_tokens":63000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":63000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"thudm/glm-4.1v-9b-thinking":{"mode":"image_generation","base_model":"thudm/glm-4.1v-9b-thinking","provider":"openrouter","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/devstral-medium":{"mode":"chat","base_model":"mistralai/devstral-medium","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/devstral-small":{"mode":"chat","base_model":"mistralai/devstral-small","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cognitivecomputations/dolphin-mistral-24b-venice-edition:free":{"mode":"chat","base_model":"cognitivecomputations/dolphin-mistral-24b-venice-edition:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"google/gemma-3n-e2b-it:free":{"mode":"chat","base_model":"google/gemma-3n-e2b-it:free","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"tencent/hunyuan-a13b-instruct:free":{"mode":"chat","base_model":"tencent/hunyuan-a13b-instruct:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"tencent/hunyuan-a13b-instruct":{"mode":"chat","base_model":"tencent/hunyuan-a13b-instruct","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"tngtech/deepseek-r1t2-chimera:free":{"mode":"chat","base_model":"tngtech/deepseek-r1t2-chimera:free","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"baidu/ernie-4.5-300b-a47b":{"mode":"chat","base_model":"baidu/ernie-4.5-300b-a47b","provider":"openrouter","max_input_tokens":123000,"max_output_tokens":123000,"max_tokens":123000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":123000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"thedrummer/anubis-70b-v1.1":{"mode":"chat","base_model":"thedrummer/anubis-70b-v1.1","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"inception/mercury":{"mode":"chat","base_model":"inception/mercury","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-small-3.2-24b-instruct:free":{"mode":"image_generation","base_model":"mistralai/mistral-small-3.2-24b-instruct:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minimax/minimax-m1":{"mode":"chat","base_model":"minimax/minimax-m1","provider":"openrouter","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-flash-lite-preview-06-17":{"mode":"image_generation","base_model":"google/gemini-2.5-flash-lite-preview-06-17","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"moonshotai/kimi-dev-72b:free":{"mode":"chat","base_model":"moonshotai/kimi-dev-72b:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/o3-pro":{"mode":"image_generation","base_model":"openai/o3-pro","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":200000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":200000,"range":{"min":1,"max":200000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"mistralai/magistral-small-2506":{"mode":"chat","base_model":"mistralai/magistral-small-2506","provider":"openrouter","max_input_tokens":40000,"max_output_tokens":40000,"max_tokens":40000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"mistralai/magistral-medium-2506":{"mode":"chat","base_model":"mistralai/magistral-medium-2506","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/magistral-medium-2506:thinking":{"mode":"chat","base_model":"mistralai/magistral-medium-2506:thinking","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-pro-preview":{"mode":"image_generation","base_model":"google/gemini-2.5-pro-preview","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-distill-qwen-7b":{"mode":"chat","base_model":"deepseek/deepseek-r1-distill-qwen-7b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-0528-qwen3-8b:free":{"mode":"chat","base_model":"deepseek/deepseek-r1-0528-qwen3-8b:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-0528-qwen3-8b":{"mode":"chat","base_model":"deepseek/deepseek-r1-0528-qwen3-8b","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-0528:free":{"mode":"chat","base_model":"deepseek/deepseek-r1-0528:free","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"sarvamai/sarvam-m:free":{"mode":"chat","base_model":"sarvamai/sarvam-m:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"thedrummer/valkyrie-49b-v1":{"mode":"chat","base_model":"thedrummer/valkyrie-49b-v1","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/devstral-small-2505:free":{"mode":"chat","base_model":"mistralai/devstral-small-2505:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/devstral-small-2505":{"mode":"chat","base_model":"mistralai/devstral-small-2505","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemma-3n-e4b-it:free":{"mode":"chat","base_model":"google/gemma-3n-e4b-it:free","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"google/gemma-3n-e4b-it":{"mode":"chat","base_model":"google/gemma-3n-e4b-it","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/codex-mini":{"mode":"image_generation","base_model":"openai/codex-mini","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"nousresearch/deephermes-3-mistral-24b-preview":{"mode":"chat","base_model":"nousresearch/deephermes-3-mistral-24b-preview","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/mistral-medium-3":{"mode":"image_generation","base_model":"mistralai/mistral-medium-3","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-pro-preview-05-06":{"mode":"image_generation","base_model":"google/gemini-2.5-pro-preview-05-06","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":1048576}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"arcee-ai/spotlight":{"mode":"image_generation","base_model":"arcee-ai/spotlight","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"arcee-ai/maestro-reasoning":{"mode":"chat","base_model":"arcee-ai/maestro-reasoning","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"arcee-ai/virtuoso-large":{"mode":"chat","base_model":"arcee-ai/virtuoso-large","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"arcee-ai/coder-large":{"mode":"chat","base_model":"arcee-ai/coder-large","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"microsoft/phi-4-reasoning-plus":{"mode":"chat","base_model":"microsoft/phi-4-reasoning-plus","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"inception/mercury-coder":{"mode":"chat","base_model":"inception/mercury-coder","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-4b:free":{"mode":"chat","base_model":"qwen/qwen3-4b:free","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"opengvlab/internvl3-14b":{"mode":"image_generation","base_model":"opengvlab/internvl3-14b","provider":"openrouter","max_input_tokens":12288,"max_output_tokens":12288,"max_tokens":12288,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":12288}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"deepseek/deepseek-prover-v2":{"mode":"chat","base_model":"deepseek/deepseek-prover-v2","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-guard-4-12b":{"mode":"image_generation","base_model":"meta-llama/llama-guard-4-12b","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen3-30b-a3b:free":{"mode":"chat","base_model":"qwen/qwen3-30b-a3b:free","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-30b-a3b":{"mode":"chat","base_model":"qwen/qwen3-30b-a3b","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-8b:free":{"mode":"chat","base_model":"qwen/qwen3-8b:free","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-8b":{"mode":"chat","base_model":"qwen/qwen3-8b","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-14b:free":{"mode":"chat","base_model":"qwen/qwen3-14b:free","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-14b":{"mode":"chat","base_model":"qwen/qwen3-14b","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-32b":{"mode":"chat","base_model":"qwen/qwen3-32b","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-235b-a22b:free":{"mode":"chat","base_model":"qwen/qwen3-235b-a22b:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen3-235b-a22b":{"mode":"chat","base_model":"qwen/qwen3-235b-a22b","provider":"openrouter","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"tngtech/deepseek-r1t-chimera:free":{"mode":"chat","base_model":"tngtech/deepseek-r1t-chimera:free","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"tngtech/deepseek-r1t-chimera":{"mode":"chat","base_model":"tngtech/deepseek-r1t-chimera","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"microsoft/mai-ds-r1:free":{"mode":"chat","base_model":"microsoft/mai-ds-r1:free","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"microsoft/mai-ds-r1":{"mode":"chat","base_model":"microsoft/mai-ds-r1","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"thudm/glm-z1-32b:free":{"mode":"chat","base_model":"thudm/glm-z1-32b:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"thudm/glm-4-32b":{"mode":"chat","base_model":"thudm/glm-4-32b","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/o4-mini-high":{"mode":"image_generation","base_model":"openai/o4-mini-high","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o3":{"mode":"image_generation","base_model":"openai/o3","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":200000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":200000,"range":{"min":1,"max":200000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o4-mini":{"mode":"image_generation","base_model":"openai/o4-mini","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":200000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":200000,"range":{"min":1,"max":200000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"shisa-ai/shisa-v2-llama3.3-70b:free":{"mode":"chat","base_model":"shisa-ai/shisa-v2-llama3.3-70b:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"shisa-ai/shisa-v2-llama3.3-70b":{"mode":"chat","base_model":"shisa-ai/shisa-v2-llama3.3-70b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"alfredpros/codellama-7b-instruct-solidity":{"mode":"chat","base_model":"alfredpros/codellama-7b-instruct-solidity","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"arliai/qwq-32b-arliai-rpr-v1:free":{"mode":"chat","base_model":"arliai/qwq-32b-arliai-rpr-v1:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"arliai/qwq-32b-arliai-rpr-v1":{"mode":"chat","base_model":"arliai/qwq-32b-arliai-rpr-v1","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"agentica-org/deepcoder-14b-preview:free":{"mode":"chat","base_model":"agentica-org/deepcoder-14b-preview:free","provider":"openrouter","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"agentica-org/deepcoder-14b-preview":{"mode":"chat","base_model":"agentica-org/deepcoder-14b-preview","provider":"openrouter","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"moonshotai/kimi-vl-a3b-thinking:free":{"mode":"image_generation","base_model":"moonshotai/kimi-vl-a3b-thinking:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"moonshotai/kimi-vl-a3b-thinking":{"mode":"image_generation","base_model":"moonshotai/kimi-vl-a3b-thinking","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"nvidia/llama-3.3-nemotron-super-49b-v1":{"mode":"chat","base_model":"nvidia/llama-3.3-nemotron-super-49b-v1","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"nvidia/llama-3.1-nemotron-ultra-253b-v1:free":{"mode":"chat","base_model":"nvidia/llama-3.1-nemotron-ultra-253b-v1:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"nvidia/llama-3.1-nemotron-ultra-253b-v1":{"mode":"chat","base_model":"nvidia/llama-3.1-nemotron-ultra-253b-v1","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-4-maverick":{"mode":"image_generation","base_model":"meta-llama/llama-4-maverick","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-4-scout":{"mode":"image_generation","base_model":"meta-llama/llama-4-scout","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-v3-base":{"mode":"chat","base_model":"deepseek/deepseek-v3-base","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"scb10x/llama3.1-typhoon2-70b-instruct":{"mode":"chat","base_model":"scb10x/llama3.1-typhoon2-70b-instruct","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"google/gemini-2.5-pro-exp-03-25":{"mode":"image_generation","base_model":"google/gemini-2.5-pro-exp-03-25","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":1048576}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwen2.5-vl-32b-instruct:free":{"mode":"image_generation","base_model":"qwen/qwen2.5-vl-32b-instruct:free","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen2.5-vl-32b-instruct":{"mode":"image_generation","base_model":"qwen/qwen2.5-vl-32b-instruct","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"deepseek/deepseek-chat-v3-0324:free":{"mode":"chat","base_model":"deepseek/deepseek-chat-v3-0324:free","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"featherless/qwerky-72b:free":{"mode":"chat","base_model":"featherless/qwerky-72b:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/o1-pro":{"mode":"image_generation","base_model":"openai/o1-pro","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":200000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":200000,"range":{"min":1,"max":200000}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-small-3.1-24b-instruct:free":{"mode":"image_generation","base_model":"mistralai/mistral-small-3.1-24b-instruct:free","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemma-3-4b-it:free":{"mode":"image_generation","base_model":"google/gemma-3-4b-it:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"ai21/jamba-1.6-large":{"mode":"chat","base_model":"ai21/jamba-1.6-large","provider":"openrouter","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"ai21/jamba-1.6-mini":{"mode":"chat","base_model":"ai21/jamba-1.6-mini","provider":"openrouter","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"google/gemma-3-12b-it:free":{"mode":"image_generation","base_model":"google/gemma-3-12b-it:free","provider":"openrouter","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"cohere/command-a":{"mode":"chat","base_model":"cohere/command-a","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/gpt-4o-mini-search-preview":{"mode":"chat","base_model":"openai/gpt-4o-mini-search-preview","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/gpt-4o-search-preview":{"mode":"chat","base_model":"openai/gpt-4o-search-preview","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"rekaai/reka-flash-3:free":{"mode":"chat","base_model":"rekaai/reka-flash-3:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"google/gemma-3-27b-it:free":{"mode":"image_generation","base_model":"google/gemma-3-27b-it:free","provider":"openrouter","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"thedrummer/anubis-pro-105b-v1":{"mode":"chat","base_model":"thedrummer/anubis-pro-105b-v1","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"thedrummer/skyfall-36b-v2":{"mode":"chat","base_model":"thedrummer/skyfall-36b-v2","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"microsoft/phi-4-multimodal-instruct":{"mode":"image_generation","base_model":"microsoft/phi-4-multimodal-instruct","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwq-32b:free":{"mode":"chat","base_model":"qwen/qwq-32b:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"qwen/qwq-32b":{"mode":"chat","base_model":"qwen/qwq-32b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"nousresearch/deephermes-3-llama-3-8b-preview:free":{"mode":"chat","base_model":"nousresearch/deephermes-3-llama-3-8b-preview:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash-lite-001":{"mode":"image_generation","base_model":"google/gemini-2.0-flash-lite-001","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3.7-sonnet:thinking":{"mode":"image_generation","base_model":"anthropic/claude-3.7-sonnet:thinking","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3.7-sonnet:beta":{"mode":"image_generation","base_model":"anthropic/claude-3.7-sonnet:beta","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"perplexity/r1-1776":{"mode":"chat","base_model":"perplexity/r1-1776","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/mistral-saba":{"mode":"chat","base_model":"mistralai/mistral-saba","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cognitivecomputations/dolphin3.0-r1-mistral-24b:free":{"mode":"chat","base_model":"cognitivecomputations/dolphin3.0-r1-mistral-24b:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"cognitivecomputations/dolphin3.0-r1-mistral-24b":{"mode":"chat","base_model":"cognitivecomputations/dolphin3.0-r1-mistral-24b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"cognitivecomputations/dolphin3.0-mistral-24b:free":{"mode":"chat","base_model":"cognitivecomputations/dolphin3.0-mistral-24b:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"cognitivecomputations/dolphin3.0-mistral-24b":{"mode":"chat","base_model":"cognitivecomputations/dolphin3.0-mistral-24b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-guard-3-8b":{"mode":"chat","base_model":"meta-llama/llama-guard-3-8b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-distill-llama-8b":{"mode":"chat","base_model":"deepseek/deepseek-r1-distill-llama-8b","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"aion-labs/aion-1.0":{"mode":"chat","base_model":"aion-labs/aion-1.0","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"aion-labs/aion-1.0-mini":{"mode":"chat","base_model":"aion-labs/aion-1.0-mini","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"aion-labs/aion-rp-llama-3.1-8b":{"mode":"chat","base_model":"aion-labs/aion-rp-llama-3.1-8b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen-vl-max":{"mode":"image_generation","base_model":"qwen/qwen-vl-max","provider":"openrouter","max_input_tokens":7500,"max_output_tokens":7500,"max_tokens":7500,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":7500}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen-turbo":{"mode":"chat","base_model":"qwen/qwen-turbo","provider":"openrouter","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen/qwen2.5-vl-72b-instruct:free":{"mode":"image_generation","base_model":"qwen/qwen2.5-vl-72b-instruct:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen2.5-vl-72b-instruct":{"mode":"image_generation","base_model":"qwen/qwen2.5-vl-72b-instruct","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen-plus":{"mode":"chat","base_model":"qwen/qwen-plus","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen/qwen-max":{"mode":"chat","base_model":"qwen/qwen-max","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-distill-qwen-1.5b":{"mode":"chat","base_model":"deepseek/deepseek-r1-distill-qwen-1.5b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-small-24b-instruct-2501:free":{"mode":"chat","base_model":"mistralai/mistral-small-24b-instruct-2501:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/mistral-small-24b-instruct-2501":{"mode":"chat","base_model":"mistralai/mistral-small-24b-instruct-2501","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-distill-qwen-32b":{"mode":"chat","base_model":"deepseek/deepseek-r1-distill-qwen-32b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-distill-qwen-14b:free":{"mode":"chat","base_model":"deepseek/deepseek-r1-distill-qwen-14b:free","provider":"openrouter","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-distill-qwen-14b":{"mode":"chat","base_model":"deepseek/deepseek-r1-distill-qwen-14b","provider":"openrouter","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"liquid/lfm-7b":{"mode":"chat","base_model":"liquid/lfm-7b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"liquid/lfm-3b":{"mode":"chat","base_model":"liquid/lfm-3b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-distill-llama-70b:free":{"mode":"chat","base_model":"deepseek/deepseek-r1-distill-llama-70b:free","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1-distill-llama-70b":{"mode":"chat","base_model":"deepseek/deepseek-r1-distill-llama-70b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek/deepseek-r1:free":{"mode":"chat","base_model":"deepseek/deepseek-r1:free","provider":"openrouter","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"minimax/minimax-01":{"mode":"image_generation","base_model":"minimax/minimax-01","provider":"openrouter","max_input_tokens":1000192,"max_output_tokens":1000192,"max_tokens":1000192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000192}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/codestral-2501":{"mode":"chat","base_model":"mistralai/codestral-2501","provider":"openrouter","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"microsoft/phi-4":{"mode":"chat","base_model":"microsoft/phi-4","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"sao10k/l3.3-euryale-70b":{"mode":"chat","base_model":"sao10k/l3.3-euryale-70b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"cohere/command-r7b-12-2024":{"mode":"chat","base_model":"cohere/command-r7b-12-2024","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash-exp:free":{"mode":"image_generation","base_model":"google/gemini-2.0-flash-exp:free","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.3-70b-instruct:free":{"mode":"chat","base_model":"meta-llama/llama-3.3-70b-instruct:free","provider":"openrouter","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"amazon/nova-lite-v1":{"mode":"image_generation","base_model":"amazon/nova-lite-v1","provider":"openrouter","max_input_tokens":300000,"max_output_tokens":300000,"max_tokens":300000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":300000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"amazon/nova-micro-v1":{"mode":"chat","base_model":"amazon/nova-micro-v1","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"amazon/nova-pro-v1":{"mode":"image_generation","base_model":"amazon/nova-pro-v1","provider":"openrouter","max_input_tokens":300000,"max_output_tokens":300000,"max_tokens":300000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":300000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen/qwq-32b-preview":{"mode":"chat","base_model":"qwen/qwq-32b-preview","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/gpt-4o-2024-11-20":{"mode":"image_generation","base_model":"openai/gpt-4o-2024-11-20","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-large-2411":{"mode":"chat","base_model":"mistralai/mistral-large-2411","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-large-2407":{"mode":"chat","base_model":"mistralai/mistral-large-2407","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/pixtral-large-2411":{"mode":"image_generation","base_model":"mistralai/pixtral-large-2411","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"infermatic/mn-inferor-12b":{"mode":"chat","base_model":"infermatic/mn-inferor-12b","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen-2.5-coder-32b-instruct:free":{"mode":"chat","base_model":"qwen/qwen-2.5-coder-32b-instruct:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"raifle/sorcererlm-8x22b":{"mode":"chat","base_model":"raifle/sorcererlm-8x22b","provider":"openrouter","max_input_tokens":16000,"max_output_tokens":16000,"max_tokens":16000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"thedrummer/unslopnemo-12b":{"mode":"chat","base_model":"thedrummer/unslopnemo-12b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthracite-org/magnum-v4-72b":{"mode":"chat","base_model":"anthracite-org/magnum-v4-72b","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/ministral-8b":{"mode":"chat","base_model":"mistralai/ministral-8b","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/ministral-3b":{"mode":"chat","base_model":"mistralai/ministral-3b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen-2.5-7b-instruct":{"mode":"chat","base_model":"qwen/qwen-2.5-7b-instruct","provider":"openrouter","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"nvidia/llama-3.1-nemotron-70b-instruct":{"mode":"chat","base_model":"nvidia/llama-3.1-nemotron-70b-instruct","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"inflection/inflection-3-productivity":{"mode":"chat","base_model":"inflection/inflection-3-productivity","provider":"openrouter","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"inflection/inflection-3-pi":{"mode":"chat","base_model":"inflection/inflection-3-pi","provider":"openrouter","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"thedrummer/rocinante-12b":{"mode":"chat","base_model":"thedrummer/rocinante-12b","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"liquid/lfm-40b":{"mode":"chat","base_model":"liquid/lfm-40b","provider":"openrouter","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"anthracite-org/magnum-v2-72b":{"mode":"chat","base_model":"anthracite-org/magnum-v2-72b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.2-3b-instruct:free":{"mode":"chat","base_model":"meta-llama/llama-3.2-3b-instruct:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.2-3b-instruct":{"mode":"chat","base_model":"meta-llama/llama-3.2-3b-instruct","provider":"openrouter","max_input_tokens":20000,"max_output_tokens":20000,"max_tokens":20000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":20000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.2-90b-vision-instruct":{"mode":"image_generation","base_model":"meta-llama/llama-3.2-90b-vision-instruct","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.2-1b-instruct":{"mode":"chat","base_model":"meta-llama/llama-3.2-1b-instruct","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.2-11b-vision-instruct:free":{"mode":"image_generation","base_model":"meta-llama/llama-3.2-11b-vision-instruct:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.2-11b-vision-instruct":{"mode":"image_generation","base_model":"meta-llama/llama-3.2-11b-vision-instruct","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen/qwen-2.5-72b-instruct:free":{"mode":"chat","base_model":"qwen/qwen-2.5-72b-instruct:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen-2.5-72b-instruct":{"mode":"chat","base_model":"qwen/qwen-2.5-72b-instruct","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neversleep/llama-3.1-lumimaid-8b":{"mode":"chat","base_model":"neversleep/llama-3.1-lumimaid-8b","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/pixtral-12b":{"mode":"image_generation","base_model":"mistralai/pixtral-12b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cohere/command-r-08-2024":{"mode":"chat","base_model":"cohere/command-r-08-2024","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cohere/command-r-plus-08-2024":{"mode":"chat","base_model":"cohere/command-r-plus-08-2024","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen/qwen-2.5-vl-7b-instruct":{"mode":"image_generation","base_model":"qwen/qwen-2.5-vl-7b-instruct","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"sao10k/l3.1-euryale-70b":{"mode":"chat","base_model":"sao10k/l3.1-euryale-70b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"microsoft/phi-3.5-mini-128k-instruct":{"mode":"chat","base_model":"microsoft/phi-3.5-mini-128k-instruct","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"nousresearch/hermes-3-llama-3.1-70b":{"mode":"chat","base_model":"nousresearch/hermes-3-llama-3.1-70b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nousresearch/hermes-3-llama-3.1-405b":{"mode":"chat","base_model":"nousresearch/hermes-3-llama-3.1-405b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/chatgpt-4o-latest":{"mode":"image_generation","base_model":"openai/chatgpt-4o-latest","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"sao10k/l3-lunaris-8b":{"mode":"chat","base_model":"sao10k/l3-lunaris-8b","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/gpt-4o-2024-08-06":{"mode":"image_generation","base_model":"openai/gpt-4o-2024-08-06","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.1-405b":{"mode":"chat","base_model":"meta-llama/llama-3.1-405b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.1-70b-instruct":{"mode":"chat","base_model":"meta-llama/llama-3.1-70b-instruct","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.1-405b-instruct:free":{"mode":"chat","base_model":"meta-llama/llama-3.1-405b-instruct:free","provider":"openrouter","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.1-405b-instruct":{"mode":"chat","base_model":"meta-llama/llama-3.1-405b-instruct","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"meta-llama/llama-3.1-8b-instruct":{"mode":"chat","base_model":"meta-llama/llama-3.1-8b-instruct","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-nemo:free":{"mode":"chat","base_model":"mistralai/mistral-nemo:free","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/mistral-nemo":{"mode":"chat","base_model":"mistralai/mistral-nemo","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o-mini-2024-07-18":{"mode":"image_generation","base_model":"openai/gpt-4o-mini-2024-07-18","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o-mini":{"mode":"image_generation","base_model":"openai/gpt-4o-mini","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemma-2-27b-it":{"mode":"chat","base_model":"google/gemma-2-27b-it","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"google/gemma-2-9b-it:free":{"mode":"chat","base_model":"google/gemma-2-9b-it:free","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"google/gemma-2-9b-it":{"mode":"chat","base_model":"google/gemma-2-9b-it","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"sao10k/l3-euryale-70b":{"mode":"chat","base_model":"sao10k/l3-euryale-70b","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"cognitivecomputations/dolphin-mixtral-8x22b":{"mode":"chat","base_model":"cognitivecomputations/dolphin-mixtral-8x22b","provider":"openrouter","max_input_tokens":16000,"max_output_tokens":16000,"max_tokens":16000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"qwen/qwen-2-72b-instruct":{"mode":"chat","base_model":"qwen/qwen-2-72b-instruct","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/mistral-7b-instruct-v0.3":{"mode":"chat","base_model":"mistralai/mistral-7b-instruct-v0.3","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nousresearch/hermes-2-pro-llama-3-8b":{"mode":"chat","base_model":"nousresearch/hermes-2-pro-llama-3-8b","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"mistralai/mistral-7b-instruct:free":{"mode":"chat","base_model":"mistralai/mistral-7b-instruct:free","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"microsoft/phi-3-mini-128k-instruct":{"mode":"chat","base_model":"microsoft/phi-3-mini-128k-instruct","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"microsoft/phi-3-medium-128k-instruct":{"mode":"chat","base_model":"microsoft/phi-3-medium-128k-instruct","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"neversleep/llama-3-lumimaid-70b":{"mode":"chat","base_model":"neversleep/llama-3-lumimaid-70b","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-guard-2-8b":{"mode":"chat","base_model":"meta-llama/llama-guard-2-8b","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/gpt-4o:extended":{"mode":"image_generation","base_model":"openai/gpt-4o:extended","provider":"openrouter","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sao10k/fimbulvetr-11b-v2":{"mode":"chat","base_model":"sao10k/fimbulvetr-11b-v2","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"meta-llama/llama-3-8b-instruct":{"mode":"chat","base_model":"meta-llama/llama-3-8b-instruct","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"microsoft/wizardlm-2-8x22b":{"mode":"chat","base_model":"microsoft/wizardlm-2-8x22b","provider":"openrouter","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/gpt-4-turbo":{"mode":"image_generation","base_model":"openai/gpt-4-turbo","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cohere/command-r-plus":{"mode":"chat","base_model":"cohere/command-r-plus","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":128000}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":128000,"step":10}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cohere/command-r-plus-04-2024":{"mode":"chat","base_model":"cohere/command-r-plus-04-2024","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sophosympatheia/midnight-rose-70b":{"mode":"chat","base_model":"sophosympatheia/midnight-rose-70b","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"cohere/command":{"mode":"chat","base_model":"cohere/command","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4096}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":4000,"step":1}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"cohere/command-r":{"mode":"chat","base_model":"cohere/command-r","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":128000}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":128000,"step":10}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3-haiku:beta":{"mode":"image_generation","base_model":"anthropic/claude-3-haiku:beta","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3-opus:beta":{"mode":"image_generation","base_model":"anthropic/claude-3-opus:beta","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3-opus":{"mode":"image_generation","base_model":"anthropic/claude-3-opus","provider":"openrouter","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cohere/command-r-03-2024":{"mode":"chat","base_model":"cohere/command-r-03-2024","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-3.5-turbo-0613":{"mode":"chat","base_model":"openai/gpt-3.5-turbo-0613","provider":"openrouter","max_input_tokens":4095,"max_output_tokens":4095,"max_tokens":4095,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4095}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4-turbo-preview":{"mode":"chat","base_model":"openai/gpt-4-turbo-preview","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-small":{"mode":"chat","base_model":"mistralai/mistral-small","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-tiny":{"mode":"chat","base_model":"mistralai/mistral-tiny","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/mixtral-8x7b-instruct":{"mode":"chat","base_model":"mistralai/mixtral-8x7b-instruct","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neversleep/noromaid-20b":{"mode":"chat","base_model":"neversleep/noromaid-20b","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"alpindale/goliath-120b":{"mode":"chat","base_model":"alpindale/goliath-120b","provider":"openrouter","max_input_tokens":6144,"max_output_tokens":6144,"max_tokens":6144,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":6144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/auto":{"mode":"chat","base_model":"openrouter/auto","provider":"openrouter","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":2000000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/gpt-4-1106-preview":{"mode":"chat","base_model":"openai/gpt-4-1106-preview","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistralai/mistral-7b-instruct-v0.1":{"mode":"chat","base_model":"mistralai/mistral-7b-instruct-v0.1","provider":"openrouter","max_input_tokens":2824,"max_output_tokens":2824,"max_tokens":2824,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2824}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-3.5-turbo-instruct":{"mode":"chat","base_model":"openai/gpt-3.5-turbo-instruct","provider":"openrouter","max_input_tokens":4095,"max_output_tokens":4095,"max_tokens":4095,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4095}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"pygmalionai/mythalion-13b":{"mode":"chat","base_model":"pygmalionai/mythalion-13b","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openai/gpt-4-0314":{"mode":"chat","base_model":"openai/gpt-4-0314","provider":"openrouter","max_input_tokens":8191,"max_output_tokens":8191,"max_tokens":8191,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen-3-235b-a22b-thinking-2507":{"mode":"chat","base_model":"qwen-3-235b-a22b-thinking-2507","provider":"cerebras","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"gpt-oss:latest":{"mode":"chat","base_model":"gpt-oss:latest","provider":"ollama","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"atla/selene-mini:latest":{"mode":"chat","base_model":"atla/selene-mini:latest","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"atla/selene-mini:fp_16":{"mode":"chat","base_model":"atla/selene-mini:fp_16","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"Osmosis/Osmosis-Structure-0.6B:latest":{"mode":"chat","base_model":"Osmosis/Osmosis-Structure-0.6B:latest","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:1b":{"mode":"chat","base_model":"gemma3:1b","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:27b":{"mode":"chat","base_model":"gemma3:27b","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:1b-it-qat":{"mode":"chat","base_model":"gemma3:1b-it-qat","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:1b-it-q4_K_M":{"mode":"chat","base_model":"gemma3:1b-it-q4_K_M","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:1b-it-q8_0":{"mode":"chat","base_model":"gemma3:1b-it-q8_0","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:1b-it-fp16":{"mode":"chat","base_model":"gemma3:1b-it-fp16","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:4b-it-qat":{"mode":"chat","base_model":"gemma3:4b-it-qat","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:4b-it-q4_K_M":{"mode":"chat","base_model":"gemma3:4b-it-q4_K_M","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:4b-it-q8_0":{"mode":"chat","base_model":"gemma3:4b-it-q8_0","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:4b-it-fp16":{"mode":"chat","base_model":"gemma3:4b-it-fp16","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:12b-it-qat":{"mode":"chat","base_model":"gemma3:12b-it-qat","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:12b-it-q4_K_M":{"mode":"chat","base_model":"gemma3:12b-it-q4_K_M","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:12b-it-q8_0":{"mode":"chat","base_model":"gemma3:12b-it-q8_0","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:12b-it-fp16":{"mode":"chat","base_model":"gemma3:12b-it-fp16","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:27b-it-qat":{"mode":"chat","base_model":"gemma3:27b-it-qat","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:27b-it-q4_K_M":{"mode":"chat","base_model":"gemma3:27b-it-q4_K_M","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:27b-it-q8_0":{"mode":"chat","base_model":"gemma3:27b-it-q8_0","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma3:27b-it-fp16":{"mode":"chat","base_model":"gemma3:27b-it-fp16","provider":"ollama","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":8192}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:0.6b-q4_K_M":{"mode":"chat","base_model":"qwen3:0.6b-q4_K_M","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:0.6b-q8_0":{"mode":"chat","base_model":"qwen3:0.6b-q8_0","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:0.6b-fp16":{"mode":"chat","base_model":"qwen3:0.6b-fp16","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:1.7b-q4_K_M":{"mode":"chat","base_model":"qwen3:1.7b-q4_K_M","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:1.7b-q8_0":{"mode":"chat","base_model":"qwen3:1.7b-q8_0","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:1.7b-fp16":{"mode":"chat","base_model":"qwen3:1.7b-fp16","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:4b-q4_K_M":{"mode":"chat","base_model":"qwen3:4b-q4_K_M","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:4b-q8_0":{"mode":"chat","base_model":"qwen3:4b-q8_0","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:4b-fp16":{"mode":"chat","base_model":"qwen3:4b-fp16","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:8b-q4_K_M":{"mode":"chat","base_model":"qwen3:8b-q4_K_M","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:8b-q8_0":{"mode":"chat","base_model":"qwen3:8b-q8_0","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:8b-fp16":{"mode":"chat","base_model":"qwen3:8b-fp16","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:14b-q4_K_M":{"mode":"chat","base_model":"qwen3:14b-q4_K_M","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:14b-q8_0":{"mode":"chat","base_model":"qwen3:14b-q8_0","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:14b-fp16":{"mode":"chat","base_model":"qwen3:14b-fp16","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:30b-a3b-q4_K_M":{"mode":"chat","base_model":"qwen3:30b-a3b-q4_K_M","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:30b-a3b-q8_0":{"mode":"chat","base_model":"qwen3:30b-a3b-q8_0","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:30b-a3b-fp16":{"mode":"chat","base_model":"qwen3:30b-a3b-fp16","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:32b-q4_K_M":{"mode":"chat","base_model":"qwen3:32b-q4_K_M","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:32b-q8_0":{"mode":"chat","base_model":"qwen3:32b-q8_0","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:32b-fp16":{"mode":"chat","base_model":"qwen3:32b-fp16","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:235b-a22b-q4_K_M":{"mode":"chat","base_model":"qwen3:235b-a22b-q4_K_M","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:235b-a22b-q8_0":{"mode":"chat","base_model":"qwen3:235b-a22b-q8_0","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen3:235b-a22b-fp16":{"mode":"chat","base_model":"qwen3:235b-a22b-fp16","provider":"ollama","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":32768}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:1.5b":{"mode":"chat","base_model":"deepseek-r1:1.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:7b":{"mode":"chat","base_model":"deepseek-r1:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:14b":{"mode":"chat","base_model":"deepseek-r1:14b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:32b":{"mode":"chat","base_model":"deepseek-r1:32b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:70b":{"mode":"chat","base_model":"deepseek-r1:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:1.5b-qwen-distill-fp16":{"mode":"chat","base_model":"deepseek-r1:1.5b-qwen-distill-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:1.5b-qwen-distill-q4_K_M":{"mode":"chat","base_model":"deepseek-r1:1.5b-qwen-distill-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:1.5b-qwen-distill-q8_0":{"mode":"chat","base_model":"deepseek-r1:1.5b-qwen-distill-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:14b-qwen-distill-fp16":{"mode":"chat","base_model":"deepseek-r1:14b-qwen-distill-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:14b-qwen-distill-q4_K_M":{"mode":"chat","base_model":"deepseek-r1:14b-qwen-distill-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:14b-qwen-distill-q8_0":{"mode":"chat","base_model":"deepseek-r1:14b-qwen-distill-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:32b-qwen-distill-fp16":{"mode":"chat","base_model":"deepseek-r1:32b-qwen-distill-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:32b-qwen-distill-q4_K_M":{"mode":"chat","base_model":"deepseek-r1:32b-qwen-distill-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:32b-qwen-distill-q8_0":{"mode":"chat","base_model":"deepseek-r1:32b-qwen-distill-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:671b-fp16":{"mode":"chat","base_model":"deepseek-r1:671b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:671b-q4_K_M":{"mode":"chat","base_model":"deepseek-r1:671b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:671b-q8_0":{"mode":"chat","base_model":"deepseek-r1:671b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:70b-llama-distill-fp16":{"mode":"chat","base_model":"deepseek-r1:70b-llama-distill-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:70b-llama-distill-q4_K_M":{"mode":"chat","base_model":"deepseek-r1:70b-llama-distill-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:70b-llama-distill-q8_0":{"mode":"chat","base_model":"deepseek-r1:70b-llama-distill-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:7b-qwen-distill-fp16":{"mode":"chat","base_model":"deepseek-r1:7b-qwen-distill-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:7b-qwen-distill-q4_K_M":{"mode":"chat","base_model":"deepseek-r1:7b-qwen-distill-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:7b-qwen-distill-q8_0":{"mode":"chat","base_model":"deepseek-r1:7b-qwen-distill-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:8b-llama-distill-fp16":{"mode":"chat","base_model":"deepseek-r1:8b-llama-distill-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:8b-llama-distill-q4_K_M":{"mode":"chat","base_model":"deepseek-r1:8b-llama-distill-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"deepseek-r1:8b-llama-distill-q8_0":{"mode":"chat","base_model":"deepseek-r1:8b-llama-distill-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-fp16":{"mode":"chat","base_model":"llama3.3:70b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q2_K":{"mode":"chat","base_model":"llama3.3:70b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q3_K_M":{"mode":"chat","base_model":"llama3.3:70b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q3_K_S":{"mode":"chat","base_model":"llama3.3:70b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q4_0":{"mode":"chat","base_model":"llama3.3:70b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3.3:70b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q4_K_S":{"mode":"chat","base_model":"llama3.3:70b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q5_0":{"mode":"chat","base_model":"llama3.3:70b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q5_1":{"mode":"chat","base_model":"llama3.3:70b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q5_K_M":{"mode":"chat","base_model":"llama3.3:70b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q6_K":{"mode":"chat","base_model":"llama3.3:70b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.3:70b-instruct-q8_0":{"mode":"chat","base_model":"llama3.3:70b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi4:14b":{"mode":"chat","base_model":"phi4:14b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi4:14b-fp16":{"mode":"chat","base_model":"phi4:14b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi4:14b-q4_K_M":{"mode":"chat","base_model":"phi4:14b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi4:14b-q8_0":{"mode":"chat","base_model":"phi4:14b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-fp16":{"mode":"chat","base_model":"llama3.2:1b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q2_K":{"mode":"chat","base_model":"llama3.2:1b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q3_K_L":{"mode":"chat","base_model":"llama3.2:1b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q3_K_M":{"mode":"chat","base_model":"llama3.2:1b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q3_K_S":{"mode":"chat","base_model":"llama3.2:1b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q4_0":{"mode":"chat","base_model":"llama3.2:1b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q4_1":{"mode":"chat","base_model":"llama3.2:1b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3.2:1b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q4_K_S":{"mode":"chat","base_model":"llama3.2:1b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q5_0":{"mode":"chat","base_model":"llama3.2:1b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q5_1":{"mode":"chat","base_model":"llama3.2:1b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q5_K_M":{"mode":"chat","base_model":"llama3.2:1b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q5_K_S":{"mode":"chat","base_model":"llama3.2:1b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q6_K":{"mode":"chat","base_model":"llama3.2:1b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-instruct-q8_0":{"mode":"chat","base_model":"llama3.2:1b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-fp16":{"mode":"chat","base_model":"llama3.2:1b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q2_K":{"mode":"chat","base_model":"llama3.2:1b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q3_K_L":{"mode":"chat","base_model":"llama3.2:1b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q3_K_M":{"mode":"chat","base_model":"llama3.2:1b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q3_K_S":{"mode":"chat","base_model":"llama3.2:1b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q4_0":{"mode":"chat","base_model":"llama3.2:1b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q4_1":{"mode":"chat","base_model":"llama3.2:1b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q4_K_M":{"mode":"chat","base_model":"llama3.2:1b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q4_K_S":{"mode":"chat","base_model":"llama3.2:1b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q5_0":{"mode":"chat","base_model":"llama3.2:1b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q5_1":{"mode":"chat","base_model":"llama3.2:1b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q5_K_M":{"mode":"chat","base_model":"llama3.2:1b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q5_K_S":{"mode":"chat","base_model":"llama3.2:1b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q6_K":{"mode":"chat","base_model":"llama3.2:1b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:1b-text-q8_0":{"mode":"chat","base_model":"llama3.2:1b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-fp16":{"mode":"chat","base_model":"llama3.2:3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q2_K":{"mode":"chat","base_model":"llama3.2:3b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q3_K_L":{"mode":"chat","base_model":"llama3.2:3b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q3_K_M":{"mode":"chat","base_model":"llama3.2:3b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q3_K_S":{"mode":"chat","base_model":"llama3.2:3b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q4_0":{"mode":"chat","base_model":"llama3.2:3b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q4_1":{"mode":"chat","base_model":"llama3.2:3b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3.2:3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q4_K_S":{"mode":"chat","base_model":"llama3.2:3b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q5_0":{"mode":"chat","base_model":"llama3.2:3b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q5_1":{"mode":"chat","base_model":"llama3.2:3b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q5_K_M":{"mode":"chat","base_model":"llama3.2:3b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q5_K_S":{"mode":"chat","base_model":"llama3.2:3b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q6_K":{"mode":"chat","base_model":"llama3.2:3b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-instruct-q8_0":{"mode":"chat","base_model":"llama3.2:3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-fp16":{"mode":"chat","base_model":"llama3.2:3b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q2_K":{"mode":"chat","base_model":"llama3.2:3b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q3_K_L":{"mode":"chat","base_model":"llama3.2:3b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q3_K_M":{"mode":"chat","base_model":"llama3.2:3b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q3_K_S":{"mode":"chat","base_model":"llama3.2:3b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q4_0":{"mode":"chat","base_model":"llama3.2:3b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q4_1":{"mode":"chat","base_model":"llama3.2:3b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q4_K_M":{"mode":"chat","base_model":"llama3.2:3b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q4_K_S":{"mode":"chat","base_model":"llama3.2:3b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q5_0":{"mode":"chat","base_model":"llama3.2:3b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q5_1":{"mode":"chat","base_model":"llama3.2:3b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q5_K_M":{"mode":"chat","base_model":"llama3.2:3b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q5_K_S":{"mode":"chat","base_model":"llama3.2:3b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q6_K":{"mode":"chat","base_model":"llama3.2:3b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2:3b-text-q8_0":{"mode":"chat","base_model":"llama3.2:3b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-fp16":{"mode":"chat","base_model":"llama3.1:405b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q2_K":{"mode":"chat","base_model":"llama3.1:405b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q3_K_L":{"mode":"chat","base_model":"llama3.1:405b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q3_K_M":{"mode":"chat","base_model":"llama3.1:405b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q3_K_S":{"mode":"chat","base_model":"llama3.1:405b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q4_0":{"mode":"chat","base_model":"llama3.1:405b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q4_1":{"mode":"chat","base_model":"llama3.1:405b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3.1:405b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q4_K_S":{"mode":"chat","base_model":"llama3.1:405b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q5_0":{"mode":"chat","base_model":"llama3.1:405b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q5_1":{"mode":"chat","base_model":"llama3.1:405b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q5_K_M":{"mode":"chat","base_model":"llama3.1:405b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q5_K_S":{"mode":"chat","base_model":"llama3.1:405b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q6_K":{"mode":"chat","base_model":"llama3.1:405b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-instruct-q8_0":{"mode":"chat","base_model":"llama3.1:405b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-fp16":{"mode":"chat","base_model":"llama3.1:405b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q2_K":{"mode":"chat","base_model":"llama3.1:405b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q3_K_L":{"mode":"chat","base_model":"llama3.1:405b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q3_K_M":{"mode":"chat","base_model":"llama3.1:405b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q3_K_S":{"mode":"chat","base_model":"llama3.1:405b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q4_0":{"mode":"chat","base_model":"llama3.1:405b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q4_1":{"mode":"chat","base_model":"llama3.1:405b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q4_K_M":{"mode":"chat","base_model":"llama3.1:405b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q4_K_S":{"mode":"chat","base_model":"llama3.1:405b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q5_0":{"mode":"chat","base_model":"llama3.1:405b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q5_1":{"mode":"chat","base_model":"llama3.1:405b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q5_K_M":{"mode":"chat","base_model":"llama3.1:405b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q5_K_S":{"mode":"chat","base_model":"llama3.1:405b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q6_K":{"mode":"chat","base_model":"llama3.1:405b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:405b-text-q8_0":{"mode":"chat","base_model":"llama3.1:405b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-fp16":{"mode":"chat","base_model":"llama3.1:70b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q2_K":{"mode":"chat","base_model":"llama3.1:70b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q3_K_L":{"mode":"chat","base_model":"llama3.1:70b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q3_K_M":{"mode":"chat","base_model":"llama3.1:70b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q3_K_S":{"mode":"chat","base_model":"llama3.1:70b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q4_0":{"mode":"chat","base_model":"llama3.1:70b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3.1:70b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q4_K_S":{"mode":"chat","base_model":"llama3.1:70b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q5_0":{"mode":"chat","base_model":"llama3.1:70b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q5_1":{"mode":"chat","base_model":"llama3.1:70b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q5_K_M":{"mode":"chat","base_model":"llama3.1:70b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q5_K_S":{"mode":"chat","base_model":"llama3.1:70b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q6_K":{"mode":"chat","base_model":"llama3.1:70b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-instruct-q8_0":{"mode":"chat","base_model":"llama3.1:70b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-fp16":{"mode":"chat","base_model":"llama3.1:70b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q2_K":{"mode":"chat","base_model":"llama3.1:70b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q3_K_L":{"mode":"chat","base_model":"llama3.1:70b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q3_K_M":{"mode":"chat","base_model":"llama3.1:70b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q3_K_S":{"mode":"chat","base_model":"llama3.1:70b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q4_0":{"mode":"chat","base_model":"llama3.1:70b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q4_1":{"mode":"chat","base_model":"llama3.1:70b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q4_K_M":{"mode":"chat","base_model":"llama3.1:70b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q4_K_S":{"mode":"chat","base_model":"llama3.1:70b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q5_0":{"mode":"chat","base_model":"llama3.1:70b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q5_1":{"mode":"chat","base_model":"llama3.1:70b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q5_K_M":{"mode":"chat","base_model":"llama3.1:70b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q5_K_S":{"mode":"chat","base_model":"llama3.1:70b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q6_K":{"mode":"chat","base_model":"llama3.1:70b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:70b-text-q8_0":{"mode":"chat","base_model":"llama3.1:70b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-fp16":{"mode":"chat","base_model":"llama3.1:8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q2_K":{"mode":"chat","base_model":"llama3.1:8b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q3_K_L":{"mode":"chat","base_model":"llama3.1:8b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q3_K_M":{"mode":"chat","base_model":"llama3.1:8b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q3_K_S":{"mode":"chat","base_model":"llama3.1:8b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q4_0":{"mode":"chat","base_model":"llama3.1:8b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q4_1":{"mode":"chat","base_model":"llama3.1:8b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3.1:8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q4_K_S":{"mode":"chat","base_model":"llama3.1:8b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q5_0":{"mode":"chat","base_model":"llama3.1:8b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q5_1":{"mode":"chat","base_model":"llama3.1:8b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q5_K_M":{"mode":"chat","base_model":"llama3.1:8b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q5_K_S":{"mode":"chat","base_model":"llama3.1:8b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q6_K":{"mode":"chat","base_model":"llama3.1:8b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-instruct-q8_0":{"mode":"chat","base_model":"llama3.1:8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-fp16":{"mode":"chat","base_model":"llama3.1:8b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q2_K":{"mode":"chat","base_model":"llama3.1:8b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q3_K_L":{"mode":"chat","base_model":"llama3.1:8b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q3_K_M":{"mode":"chat","base_model":"llama3.1:8b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q3_K_S":{"mode":"chat","base_model":"llama3.1:8b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q4_0":{"mode":"chat","base_model":"llama3.1:8b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q4_1":{"mode":"chat","base_model":"llama3.1:8b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q4_K_M":{"mode":"chat","base_model":"llama3.1:8b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q4_K_S":{"mode":"chat","base_model":"llama3.1:8b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q5_0":{"mode":"chat","base_model":"llama3.1:8b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q5_1":{"mode":"chat","base_model":"llama3.1:8b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q5_K_M":{"mode":"chat","base_model":"llama3.1:8b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q5_K_S":{"mode":"chat","base_model":"llama3.1:8b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q6_K":{"mode":"chat","base_model":"llama3.1:8b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.1:8b-text-q8_0":{"mode":"chat","base_model":"llama3.1:8b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nomic-embed-text:137m-v1.5-fp16":{"mode":"chat","base_model":"nomic-embed-text:137m-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nomic-embed-text:v1.5":{"mode":"chat","base_model":"nomic-embed-text:v1.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-fp16":{"mode":"chat","base_model":"mistral:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q2_K":{"mode":"chat","base_model":"mistral:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q3_K_L":{"mode":"chat","base_model":"mistral:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q3_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q3_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q4_0":{"mode":"chat","base_model":"mistral:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q4_1":{"mode":"chat","base_model":"mistral:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q4_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q4_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q5_0":{"mode":"chat","base_model":"mistral:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q5_1":{"mode":"chat","base_model":"mistral:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q5_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q5_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q6_K":{"mode":"chat","base_model":"mistral:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-q8_0":{"mode":"chat","base_model":"mistral:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-fp16":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q2_K":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q3_K_L":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q3_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q3_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q4_0":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q4_1":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q4_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q4_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q5_0":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q5_1":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q5_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q5_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q6_K":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.2-q8_0":{"mode":"chat","base_model":"mistral:7b-instruct-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-fp16":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q2_K":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q3_K_L":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q3_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q3_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q4_0":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q4_1":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q4_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q4_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q5_0":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q5_1":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q5_K_M":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q5_K_S":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q6_K":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-instruct-v0.3-q8_0":{"mode":"chat","base_model":"mistral:7b-instruct-v0.3-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text":{"mode":"chat","base_model":"mistral:7b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-fp16":{"mode":"chat","base_model":"mistral:7b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q2_K":{"mode":"chat","base_model":"mistral:7b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q3_K_L":{"mode":"chat","base_model":"mistral:7b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q3_K_M":{"mode":"chat","base_model":"mistral:7b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q3_K_S":{"mode":"chat","base_model":"mistral:7b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q4_0":{"mode":"chat","base_model":"mistral:7b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q4_1":{"mode":"chat","base_model":"mistral:7b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q4_K_M":{"mode":"chat","base_model":"mistral:7b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q4_K_S":{"mode":"chat","base_model":"mistral:7b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q5_0":{"mode":"chat","base_model":"mistral:7b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q5_1":{"mode":"chat","base_model":"mistral:7b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q5_K_M":{"mode":"chat","base_model":"mistral:7b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q5_K_S":{"mode":"chat","base_model":"mistral:7b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q6_K":{"mode":"chat","base_model":"mistral:7b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-q8_0":{"mode":"chat","base_model":"mistral:7b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-fp16":{"mode":"chat","base_model":"mistral:7b-text-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q2_K":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q3_K_L":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q3_K_M":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q3_K_S":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q4_0":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q4_1":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q4_K_M":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q4_K_S":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q5_0":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q5_1":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q5_K_M":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q5_K_S":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q6_K":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:7b-text-v0.2-q8_0":{"mode":"chat","base_model":"mistral:7b-text-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:instruct":{"mode":"chat","base_model":"mistral:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:text":{"mode":"chat","base_model":"mistral:text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:v0.1":{"mode":"chat","base_model":"mistral:v0.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:v0.2":{"mode":"chat","base_model":"mistral:v0.2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral:v0.3":{"mode":"chat","base_model":"mistral:v0.3","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-fp16":{"mode":"chat","base_model":"llama3:70b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q2_K":{"mode":"chat","base_model":"llama3:70b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q3_K_L":{"mode":"chat","base_model":"llama3:70b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q3_K_M":{"mode":"chat","base_model":"llama3:70b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q3_K_S":{"mode":"chat","base_model":"llama3:70b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q4_0":{"mode":"chat","base_model":"llama3:70b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q4_1":{"mode":"chat","base_model":"llama3:70b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3:70b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q4_K_S":{"mode":"chat","base_model":"llama3:70b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q5_0":{"mode":"chat","base_model":"llama3:70b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q5_1":{"mode":"chat","base_model":"llama3:70b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q5_K_M":{"mode":"chat","base_model":"llama3:70b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q5_K_S":{"mode":"chat","base_model":"llama3:70b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q6_K":{"mode":"chat","base_model":"llama3:70b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-instruct-q8_0":{"mode":"chat","base_model":"llama3:70b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text":{"mode":"chat","base_model":"llama3:70b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-fp16":{"mode":"chat","base_model":"llama3:70b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q2_K":{"mode":"chat","base_model":"llama3:70b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q3_K_L":{"mode":"chat","base_model":"llama3:70b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q3_K_M":{"mode":"chat","base_model":"llama3:70b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q3_K_S":{"mode":"chat","base_model":"llama3:70b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q4_0":{"mode":"chat","base_model":"llama3:70b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q4_1":{"mode":"chat","base_model":"llama3:70b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q4_K_M":{"mode":"chat","base_model":"llama3:70b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q4_K_S":{"mode":"chat","base_model":"llama3:70b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q5_0":{"mode":"chat","base_model":"llama3:70b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q5_1":{"mode":"chat","base_model":"llama3:70b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q5_K_M":{"mode":"chat","base_model":"llama3:70b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q5_K_S":{"mode":"chat","base_model":"llama3:70b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q6_K":{"mode":"chat","base_model":"llama3:70b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:70b-text-q8_0":{"mode":"chat","base_model":"llama3:70b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-fp16":{"mode":"chat","base_model":"llama3:8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q2_K":{"mode":"chat","base_model":"llama3:8b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q3_K_L":{"mode":"chat","base_model":"llama3:8b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q3_K_M":{"mode":"chat","base_model":"llama3:8b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q3_K_S":{"mode":"chat","base_model":"llama3:8b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q4_0":{"mode":"chat","base_model":"llama3:8b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q4_1":{"mode":"chat","base_model":"llama3:8b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3:8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q4_K_S":{"mode":"chat","base_model":"llama3:8b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q5_0":{"mode":"chat","base_model":"llama3:8b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q5_1":{"mode":"chat","base_model":"llama3:8b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q5_K_M":{"mode":"chat","base_model":"llama3:8b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q5_K_S":{"mode":"chat","base_model":"llama3:8b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q6_K":{"mode":"chat","base_model":"llama3:8b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-instruct-q8_0":{"mode":"chat","base_model":"llama3:8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text":{"mode":"chat","base_model":"llama3:8b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-fp16":{"mode":"chat","base_model":"llama3:8b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q2_K":{"mode":"chat","base_model":"llama3:8b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q3_K_L":{"mode":"chat","base_model":"llama3:8b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q3_K_M":{"mode":"chat","base_model":"llama3:8b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q3_K_S":{"mode":"chat","base_model":"llama3:8b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q4_0":{"mode":"chat","base_model":"llama3:8b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q4_1":{"mode":"chat","base_model":"llama3:8b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q4_K_M":{"mode":"chat","base_model":"llama3:8b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q4_K_S":{"mode":"chat","base_model":"llama3:8b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q5_0":{"mode":"chat","base_model":"llama3:8b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q5_1":{"mode":"chat","base_model":"llama3:8b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q5_K_M":{"mode":"chat","base_model":"llama3:8b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q5_K_S":{"mode":"chat","base_model":"llama3:8b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q6_K":{"mode":"chat","base_model":"llama3:8b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:8b-text-q8_0":{"mode":"chat","base_model":"llama3:8b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:instruct":{"mode":"chat","base_model":"llama3:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3:text":{"mode":"chat","base_model":"llama3:text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b":{"mode":"chat","base_model":"qwen2.5:0.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b":{"mode":"chat","base_model":"qwen2.5:1.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b":{"mode":"chat","base_model":"qwen2.5:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base":{"mode":"chat","base_model":"qwen2.5:0.5b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q2_K":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q3_K_L":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q3_K_M":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q3_K_S":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q4_0":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q4_1":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q4_K_M":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q4_K_S":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q5_0":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q5_1":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q5_K_S":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-base-q8_0":{"mode":"chat","base_model":"qwen2.5:0.5b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:0.5b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5:0.5b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:1.5b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5:1.5b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5:14b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:14b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5:14b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5:32b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:32b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5:32b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct":{"mode":"chat","base_model":"qwen2.5:3b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5:3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:3b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5:3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5:72b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:72b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5:72b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5:7b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b":{"mode":"chat","base_model":"qwen:0.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b":{"mode":"chat","base_model":"qwen:1.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b":{"mode":"chat","base_model":"qwen:4b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b":{"mode":"chat","base_model":"qwen:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b":{"mode":"chat","base_model":"qwen:14b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b":{"mode":"chat","base_model":"qwen:32b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b":{"mode":"chat","base_model":"qwen:72b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b":{"mode":"chat","base_model":"qwen:110b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat":{"mode":"chat","base_model":"qwen:0.5b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-fp16":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q2_K":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q4_0":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q4_1":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q5_0":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q5_1":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q6_K":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-chat-v1.5-q8_0":{"mode":"chat","base_model":"qwen:0.5b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text":{"mode":"chat","base_model":"qwen:0.5b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-fp16":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q2_K":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q4_0":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q4_1":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q5_0":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q5_1":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q6_K":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:0.5b-text-v1.5-q8_0":{"mode":"chat","base_model":"qwen:0.5b-text-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat":{"mode":"chat","base_model":"qwen:1.8b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-fp16":{"mode":"chat","base_model":"qwen:1.8b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q2_K":{"mode":"chat","base_model":"qwen:1.8b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q3_K_L":{"mode":"chat","base_model":"qwen:1.8b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q3_K_M":{"mode":"chat","base_model":"qwen:1.8b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q3_K_S":{"mode":"chat","base_model":"qwen:1.8b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q4_0":{"mode":"chat","base_model":"qwen:1.8b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q4_1":{"mode":"chat","base_model":"qwen:1.8b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q4_K_M":{"mode":"chat","base_model":"qwen:1.8b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q4_K_S":{"mode":"chat","base_model":"qwen:1.8b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q5_0":{"mode":"chat","base_model":"qwen:1.8b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q5_1":{"mode":"chat","base_model":"qwen:1.8b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q5_K_M":{"mode":"chat","base_model":"qwen:1.8b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q5_K_S":{"mode":"chat","base_model":"qwen:1.8b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q6_K":{"mode":"chat","base_model":"qwen:1.8b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-q8_0":{"mode":"chat","base_model":"qwen:1.8b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-fp16":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q2_K":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q4_0":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q4_1":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q5_0":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q5_1":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q6_K":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-chat-v1.5-q8_0":{"mode":"chat","base_model":"qwen:1.8b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text":{"mode":"chat","base_model":"qwen:1.8b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-fp16":{"mode":"chat","base_model":"qwen:1.8b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q2_K":{"mode":"chat","base_model":"qwen:1.8b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q3_K_L":{"mode":"chat","base_model":"qwen:1.8b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q3_K_M":{"mode":"chat","base_model":"qwen:1.8b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q3_K_S":{"mode":"chat","base_model":"qwen:1.8b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q4_0":{"mode":"chat","base_model":"qwen:1.8b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q4_1":{"mode":"chat","base_model":"qwen:1.8b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q4_K_M":{"mode":"chat","base_model":"qwen:1.8b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q4_K_S":{"mode":"chat","base_model":"qwen:1.8b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q5_0":{"mode":"chat","base_model":"qwen:1.8b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q5_1":{"mode":"chat","base_model":"qwen:1.8b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q5_K_M":{"mode":"chat","base_model":"qwen:1.8b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q5_K_S":{"mode":"chat","base_model":"qwen:1.8b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q6_K":{"mode":"chat","base_model":"qwen:1.8b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-q8_0":{"mode":"chat","base_model":"qwen:1.8b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-fp16":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q2_K":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q4_0":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q4_1":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q5_0":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q5_1":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q6_K":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:1.8b-text-v1.5-q8_0":{"mode":"chat","base_model":"qwen:1.8b-text-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat":{"mode":"chat","base_model":"qwen:110b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-fp16":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q2_K":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q4_0":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q4_1":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q5_0":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q5_1":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q6_K":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-chat-v1.5-q8_0":{"mode":"chat","base_model":"qwen:110b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-fp16":{"mode":"chat","base_model":"qwen:110b-text-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q2_K":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q4_0":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q4_1":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q5_0":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q5_1":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q6_K":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:110b-text-v1.5-q8_0":{"mode":"chat","base_model":"qwen:110b-text-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat":{"mode":"chat","base_model":"qwen:14b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-fp16":{"mode":"chat","base_model":"qwen:14b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q2_K":{"mode":"chat","base_model":"qwen:14b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q3_K_L":{"mode":"chat","base_model":"qwen:14b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q3_K_M":{"mode":"chat","base_model":"qwen:14b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q3_K_S":{"mode":"chat","base_model":"qwen:14b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q4_0":{"mode":"chat","base_model":"qwen:14b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q4_1":{"mode":"chat","base_model":"qwen:14b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q4_K_M":{"mode":"chat","base_model":"qwen:14b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q4_K_S":{"mode":"chat","base_model":"qwen:14b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q5_0":{"mode":"chat","base_model":"qwen:14b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q5_1":{"mode":"chat","base_model":"qwen:14b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q5_K_M":{"mode":"chat","base_model":"qwen:14b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q5_K_S":{"mode":"chat","base_model":"qwen:14b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q6_K":{"mode":"chat","base_model":"qwen:14b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-q8_0":{"mode":"chat","base_model":"qwen:14b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-fp16":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q2_K":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q4_0":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q4_1":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q5_0":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q5_1":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q6_K":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-chat-v1.5-q8_0":{"mode":"chat","base_model":"qwen:14b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text":{"mode":"chat","base_model":"qwen:14b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-fp16":{"mode":"chat","base_model":"qwen:14b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q2_K":{"mode":"chat","base_model":"qwen:14b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q3_K_L":{"mode":"chat","base_model":"qwen:14b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q3_K_M":{"mode":"chat","base_model":"qwen:14b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q3_K_S":{"mode":"chat","base_model":"qwen:14b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q4_0":{"mode":"chat","base_model":"qwen:14b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q4_1":{"mode":"chat","base_model":"qwen:14b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q4_K_M":{"mode":"chat","base_model":"qwen:14b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q4_K_S":{"mode":"chat","base_model":"qwen:14b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q5_0":{"mode":"chat","base_model":"qwen:14b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q5_1":{"mode":"chat","base_model":"qwen:14b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q5_K_M":{"mode":"chat","base_model":"qwen:14b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q5_K_S":{"mode":"chat","base_model":"qwen:14b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q6_K":{"mode":"chat","base_model":"qwen:14b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-q8_0":{"mode":"chat","base_model":"qwen:14b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-fp16":{"mode":"chat","base_model":"qwen:14b-text-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q2_K":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q4_0":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q4_1":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q5_0":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q5_1":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q6_K":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:14b-text-v1.5-q8_0":{"mode":"chat","base_model":"qwen:14b-text-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat":{"mode":"chat","base_model":"qwen:32b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-fp16":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q2_K":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q4_0":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q4_1":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q5_0":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q5_1":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q6_K":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-chat-v1.5-q8_0":{"mode":"chat","base_model":"qwen:32b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text":{"mode":"chat","base_model":"qwen:32b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q2_K":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q4_0":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q4_1":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q5_0":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q5_1":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:32b-text-v1.5-q8_0":{"mode":"chat","base_model":"qwen:32b-text-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat":{"mode":"chat","base_model":"qwen:4b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-fp16":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q2_K":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q4_0":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q4_1":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q5_0":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q5_1":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q6_K":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-chat-v1.5-q8_0":{"mode":"chat","base_model":"qwen:4b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text":{"mode":"chat","base_model":"qwen:4b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-fp16":{"mode":"chat","base_model":"qwen:4b-text-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q2_K":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q4_0":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q4_1":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q5_0":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q5_1":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q6_K":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:4b-text-v1.5-q8_0":{"mode":"chat","base_model":"qwen:4b-text-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat":{"mode":"chat","base_model":"qwen:72b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-fp16":{"mode":"chat","base_model":"qwen:72b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q2_K":{"mode":"chat","base_model":"qwen:72b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q3_K_L":{"mode":"chat","base_model":"qwen:72b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q3_K_M":{"mode":"chat","base_model":"qwen:72b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q3_K_S":{"mode":"chat","base_model":"qwen:72b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q4_0":{"mode":"chat","base_model":"qwen:72b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q4_1":{"mode":"chat","base_model":"qwen:72b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q4_K_M":{"mode":"chat","base_model":"qwen:72b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q4_K_S":{"mode":"chat","base_model":"qwen:72b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q5_0":{"mode":"chat","base_model":"qwen:72b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q5_1":{"mode":"chat","base_model":"qwen:72b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q5_K_M":{"mode":"chat","base_model":"qwen:72b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q5_K_S":{"mode":"chat","base_model":"qwen:72b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q6_K":{"mode":"chat","base_model":"qwen:72b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-q8_0":{"mode":"chat","base_model":"qwen:72b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-fp16":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q2_K":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q4_0":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q4_1":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q5_0":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q5_1":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q6_K":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-chat-v1.5-q8_0":{"mode":"chat","base_model":"qwen:72b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text":{"mode":"chat","base_model":"qwen:72b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-fp16":{"mode":"chat","base_model":"qwen:72b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q2_K":{"mode":"chat","base_model":"qwen:72b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q3_K_L":{"mode":"chat","base_model":"qwen:72b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q3_K_M":{"mode":"chat","base_model":"qwen:72b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q3_K_S":{"mode":"chat","base_model":"qwen:72b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q4_0":{"mode":"chat","base_model":"qwen:72b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q4_1":{"mode":"chat","base_model":"qwen:72b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q4_K_M":{"mode":"chat","base_model":"qwen:72b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q4_K_S":{"mode":"chat","base_model":"qwen:72b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q5_0":{"mode":"chat","base_model":"qwen:72b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q5_1":{"mode":"chat","base_model":"qwen:72b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q5_K_M":{"mode":"chat","base_model":"qwen:72b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q5_K_S":{"mode":"chat","base_model":"qwen:72b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q6_K":{"mode":"chat","base_model":"qwen:72b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-q8_0":{"mode":"chat","base_model":"qwen:72b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-fp16":{"mode":"chat","base_model":"qwen:72b-text-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q2_K":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q4_0":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q4_1":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q5_0":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q5_1":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q6_K":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:72b-text-v1.5-q8_0":{"mode":"chat","base_model":"qwen:72b-text-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat":{"mode":"chat","base_model":"qwen:7b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-fp16":{"mode":"chat","base_model":"qwen:7b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q2_K":{"mode":"chat","base_model":"qwen:7b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q3_K_L":{"mode":"chat","base_model":"qwen:7b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q3_K_M":{"mode":"chat","base_model":"qwen:7b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q3_K_S":{"mode":"chat","base_model":"qwen:7b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q4_0":{"mode":"chat","base_model":"qwen:7b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q4_1":{"mode":"chat","base_model":"qwen:7b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q4_K_M":{"mode":"chat","base_model":"qwen:7b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q4_K_S":{"mode":"chat","base_model":"qwen:7b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q5_0":{"mode":"chat","base_model":"qwen:7b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q5_1":{"mode":"chat","base_model":"qwen:7b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q5_K_M":{"mode":"chat","base_model":"qwen:7b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q5_K_S":{"mode":"chat","base_model":"qwen:7b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q6_K":{"mode":"chat","base_model":"qwen:7b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-q8_0":{"mode":"chat","base_model":"qwen:7b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-fp16":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q2_K":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q4_0":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q4_1":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q5_0":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q5_1":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q6_K":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-chat-v1.5-q8_0":{"mode":"chat","base_model":"qwen:7b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-fp16":{"mode":"chat","base_model":"qwen:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q2_K":{"mode":"chat","base_model":"qwen:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q3_K_L":{"mode":"chat","base_model":"qwen:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q3_K_M":{"mode":"chat","base_model":"qwen:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q3_K_S":{"mode":"chat","base_model":"qwen:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q4_0":{"mode":"chat","base_model":"qwen:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q4_1":{"mode":"chat","base_model":"qwen:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q4_K_M":{"mode":"chat","base_model":"qwen:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q4_K_S":{"mode":"chat","base_model":"qwen:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q5_0":{"mode":"chat","base_model":"qwen:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q5_1":{"mode":"chat","base_model":"qwen:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q5_K_M":{"mode":"chat","base_model":"qwen:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q5_K_S":{"mode":"chat","base_model":"qwen:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q6_K":{"mode":"chat","base_model":"qwen:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-q8_0":{"mode":"chat","base_model":"qwen:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text":{"mode":"chat","base_model":"qwen:7b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-fp16":{"mode":"chat","base_model":"qwen:7b-text-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q2_K":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q3_K_L":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q3_K_M":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q3_K_S":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q4_0":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q4_1":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q4_K_M":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q4_K_S":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q5_0":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q5_1":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q5_K_M":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q5_K_S":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q6_K":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen:7b-text-v1.5-q8_0":{"mode":"chat","base_model":"qwen:7b-text-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b":{"mode":"chat","base_model":"gemma:2b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct":{"mode":"chat","base_model":"gemma:2b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-fp16":{"mode":"chat","base_model":"gemma:2b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q2_K":{"mode":"chat","base_model":"gemma:2b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q3_K_L":{"mode":"chat","base_model":"gemma:2b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q3_K_M":{"mode":"chat","base_model":"gemma:2b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q3_K_S":{"mode":"chat","base_model":"gemma:2b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q4_0":{"mode":"chat","base_model":"gemma:2b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q4_1":{"mode":"chat","base_model":"gemma:2b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q4_K_M":{"mode":"chat","base_model":"gemma:2b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q4_K_S":{"mode":"chat","base_model":"gemma:2b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q5_0":{"mode":"chat","base_model":"gemma:2b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q5_1":{"mode":"chat","base_model":"gemma:2b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q5_K_M":{"mode":"chat","base_model":"gemma:2b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q5_K_S":{"mode":"chat","base_model":"gemma:2b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q6_K":{"mode":"chat","base_model":"gemma:2b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-q8_0":{"mode":"chat","base_model":"gemma:2b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-fp16":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q2_K":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q3_K_L":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q3_K_M":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q3_K_S":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q4_0":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q4_1":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q4_K_M":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q4_K_S":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q5_0":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q5_1":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q5_K_M":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q5_K_S":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q6_K":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-instruct-v1.1-q8_0":{"mode":"chat","base_model":"gemma:2b-instruct-v1.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text":{"mode":"chat","base_model":"gemma:2b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-fp16":{"mode":"chat","base_model":"gemma:2b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q2_K":{"mode":"chat","base_model":"gemma:2b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q3_K_L":{"mode":"chat","base_model":"gemma:2b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q3_K_M":{"mode":"chat","base_model":"gemma:2b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q3_K_S":{"mode":"chat","base_model":"gemma:2b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q4_0":{"mode":"chat","base_model":"gemma:2b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q4_1":{"mode":"chat","base_model":"gemma:2b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q4_K_M":{"mode":"chat","base_model":"gemma:2b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q4_K_S":{"mode":"chat","base_model":"gemma:2b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q5_0":{"mode":"chat","base_model":"gemma:2b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q5_1":{"mode":"chat","base_model":"gemma:2b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q5_K_M":{"mode":"chat","base_model":"gemma:2b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q5_K_S":{"mode":"chat","base_model":"gemma:2b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q6_K":{"mode":"chat","base_model":"gemma:2b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-text-q8_0":{"mode":"chat","base_model":"gemma:2b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:2b-v1.1":{"mode":"chat","base_model":"gemma:2b-v1.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct":{"mode":"chat","base_model":"gemma:7b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-fp16":{"mode":"chat","base_model":"gemma:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q2_K":{"mode":"chat","base_model":"gemma:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q3_K_L":{"mode":"chat","base_model":"gemma:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q3_K_M":{"mode":"chat","base_model":"gemma:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q3_K_S":{"mode":"chat","base_model":"gemma:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q4_0":{"mode":"chat","base_model":"gemma:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q4_1":{"mode":"chat","base_model":"gemma:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q4_K_M":{"mode":"chat","base_model":"gemma:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q4_K_S":{"mode":"chat","base_model":"gemma:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q5_0":{"mode":"chat","base_model":"gemma:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q5_1":{"mode":"chat","base_model":"gemma:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q5_K_M":{"mode":"chat","base_model":"gemma:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q5_K_S":{"mode":"chat","base_model":"gemma:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q6_K":{"mode":"chat","base_model":"gemma:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-q8_0":{"mode":"chat","base_model":"gemma:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-fp16":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q2_K":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q3_K_L":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q3_K_M":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q3_K_S":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q4_0":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q4_1":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q4_K_M":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q4_K_S":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q5_0":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q5_1":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q5_K_M":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q5_K_S":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q6_K":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-instruct-v1.1-q8_0":{"mode":"chat","base_model":"gemma:7b-instruct-v1.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text":{"mode":"chat","base_model":"gemma:7b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-fp16":{"mode":"chat","base_model":"gemma:7b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q2_K":{"mode":"chat","base_model":"gemma:7b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q3_K_L":{"mode":"chat","base_model":"gemma:7b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q3_K_M":{"mode":"chat","base_model":"gemma:7b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q3_K_S":{"mode":"chat","base_model":"gemma:7b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q4_0":{"mode":"chat","base_model":"gemma:7b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q4_1":{"mode":"chat","base_model":"gemma:7b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q4_K_M":{"mode":"chat","base_model":"gemma:7b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q4_K_S":{"mode":"chat","base_model":"gemma:7b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q5_0":{"mode":"chat","base_model":"gemma:7b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q5_1":{"mode":"chat","base_model":"gemma:7b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q5_K_M":{"mode":"chat","base_model":"gemma:7b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q5_K_S":{"mode":"chat","base_model":"gemma:7b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q6_K":{"mode":"chat","base_model":"gemma:7b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-text-q8_0":{"mode":"chat","base_model":"gemma:7b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:7b-v1.1":{"mode":"chat","base_model":"gemma:7b-v1.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:instruct":{"mode":"chat","base_model":"gemma:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:text":{"mode":"chat","base_model":"gemma:text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma:v1.1":{"mode":"chat","base_model":"gemma:v1.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b":{"mode":"chat","base_model":"qwen2:0.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b":{"mode":"chat","base_model":"qwen2:1.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b":{"mode":"chat","base_model":"qwen2:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b":{"mode":"chat","base_model":"qwen2:72b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct":{"mode":"chat","base_model":"qwen2:0.5b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-fp16":{"mode":"chat","base_model":"qwen2:0.5b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q2_K":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q4_0":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q4_1":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q5_0":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q5_1":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q6_K":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:0.5b-instruct-q8_0":{"mode":"chat","base_model":"qwen2:0.5b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct":{"mode":"chat","base_model":"qwen2:1.5b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-fp16":{"mode":"chat","base_model":"qwen2:1.5b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q2_K":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q4_0":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q4_1":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q5_0":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q5_1":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q6_K":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:1.5b-instruct-q8_0":{"mode":"chat","base_model":"qwen2:1.5b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-fp16":{"mode":"chat","base_model":"qwen2:72b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q2_K":{"mode":"chat","base_model":"qwen2:72b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2:72b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2:72b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2:72b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q4_0":{"mode":"chat","base_model":"qwen2:72b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q4_1":{"mode":"chat","base_model":"qwen2:72b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2:72b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2:72b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q5_0":{"mode":"chat","base_model":"qwen2:72b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q5_1":{"mode":"chat","base_model":"qwen2:72b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2:72b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2:72b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q6_K":{"mode":"chat","base_model":"qwen2:72b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-instruct-q8_0":{"mode":"chat","base_model":"qwen2:72b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text":{"mode":"chat","base_model":"qwen2:72b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-fp16":{"mode":"chat","base_model":"qwen2:72b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q2_K":{"mode":"chat","base_model":"qwen2:72b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q3_K_L":{"mode":"chat","base_model":"qwen2:72b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q3_K_M":{"mode":"chat","base_model":"qwen2:72b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q3_K_S":{"mode":"chat","base_model":"qwen2:72b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q4_0":{"mode":"chat","base_model":"qwen2:72b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q4_1":{"mode":"chat","base_model":"qwen2:72b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q4_K_M":{"mode":"chat","base_model":"qwen2:72b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q4_K_S":{"mode":"chat","base_model":"qwen2:72b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q5_0":{"mode":"chat","base_model":"qwen2:72b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q5_1":{"mode":"chat","base_model":"qwen2:72b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q5_K_M":{"mode":"chat","base_model":"qwen2:72b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q5_K_S":{"mode":"chat","base_model":"qwen2:72b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q6_K":{"mode":"chat","base_model":"qwen2:72b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:72b-text-q8_0":{"mode":"chat","base_model":"qwen2:72b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-fp16":{"mode":"chat","base_model":"qwen2:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q2_K":{"mode":"chat","base_model":"qwen2:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q4_0":{"mode":"chat","base_model":"qwen2:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q4_1":{"mode":"chat","base_model":"qwen2:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q5_0":{"mode":"chat","base_model":"qwen2:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q5_1":{"mode":"chat","base_model":"qwen2:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q6_K":{"mode":"chat","base_model":"qwen2:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-instruct-q8_0":{"mode":"chat","base_model":"qwen2:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text":{"mode":"chat","base_model":"qwen2:7b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q2_K":{"mode":"chat","base_model":"qwen2:7b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q3_K_L":{"mode":"chat","base_model":"qwen2:7b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q3_K_M":{"mode":"chat","base_model":"qwen2:7b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q3_K_S":{"mode":"chat","base_model":"qwen2:7b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q4_0":{"mode":"chat","base_model":"qwen2:7b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q4_1":{"mode":"chat","base_model":"qwen2:7b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q4_K_M":{"mode":"chat","base_model":"qwen2:7b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q4_K_S":{"mode":"chat","base_model":"qwen2:7b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q5_0":{"mode":"chat","base_model":"qwen2:7b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q5_1":{"mode":"chat","base_model":"qwen2:7b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2:7b-text-q8_0":{"mode":"chat","base_model":"qwen2:7b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-fp16":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-base-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:0.5b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:0.5b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-fp16":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-base-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:1.5b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:1.5b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base":{"mode":"chat","base_model":"qwen2.5-coder:14b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-fp16":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-base-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:14b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:14b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:14b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base":{"mode":"chat","base_model":"qwen2.5-coder:32b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-fp16":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-base-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:32b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:32b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:32b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base":{"mode":"chat","base_model":"qwen2.5-coder:3b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-fp16":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-base-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:3b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:3b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base":{"mode":"chat","base_model":"qwen2.5-coder:7b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-fp16":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-base-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:7b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-fp16":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q2_K":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q4_0":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q4_1":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q5_0":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q5_1":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q6_K":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2.5-coder:7b-instruct-q8_0":{"mode":"chat","base_model":"qwen2.5-coder:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b":{"mode":"chat","base_model":"llava:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b":{"mode":"chat","base_model":"llava:34b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-fp16":{"mode":"chat","base_model":"llava:13b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q2_K":{"mode":"chat","base_model":"llava:13b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q3_K_L":{"mode":"chat","base_model":"llava:13b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q3_K_M":{"mode":"chat","base_model":"llava:13b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q3_K_S":{"mode":"chat","base_model":"llava:13b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q4_0":{"mode":"chat","base_model":"llava:13b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q4_1":{"mode":"chat","base_model":"llava:13b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q4_K_M":{"mode":"chat","base_model":"llava:13b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q4_K_S":{"mode":"chat","base_model":"llava:13b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q5_0":{"mode":"chat","base_model":"llava:13b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q5_1":{"mode":"chat","base_model":"llava:13b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q5_K_M":{"mode":"chat","base_model":"llava:13b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q5_K_S":{"mode":"chat","base_model":"llava:13b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q6_K":{"mode":"chat","base_model":"llava:13b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.5-q8_0":{"mode":"chat","base_model":"llava:13b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6":{"mode":"chat","base_model":"llava:13b-v1.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-fp16":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q2_K":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q3_K_L":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q3_K_M":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q3_K_S":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q4_0":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q4_1":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q4_K_M":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q4_K_S":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q5_0":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q5_1":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q5_K_M":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q5_K_S":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q6_K":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:13b-v1.6-vicuna-q8_0":{"mode":"chat","base_model":"llava:13b-v1.6-vicuna-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6":{"mode":"chat","base_model":"llava:34b-v1.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-fp16":{"mode":"chat","base_model":"llava:34b-v1.6-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q2_K":{"mode":"chat","base_model":"llava:34b-v1.6-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q3_K_L":{"mode":"chat","base_model":"llava:34b-v1.6-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q3_K_M":{"mode":"chat","base_model":"llava:34b-v1.6-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q3_K_S":{"mode":"chat","base_model":"llava:34b-v1.6-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q4_0":{"mode":"chat","base_model":"llava:34b-v1.6-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q4_1":{"mode":"chat","base_model":"llava:34b-v1.6-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q4_K_M":{"mode":"chat","base_model":"llava:34b-v1.6-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q4_K_S":{"mode":"chat","base_model":"llava:34b-v1.6-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q5_0":{"mode":"chat","base_model":"llava:34b-v1.6-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q5_1":{"mode":"chat","base_model":"llava:34b-v1.6-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q5_K_M":{"mode":"chat","base_model":"llava:34b-v1.6-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q5_K_S":{"mode":"chat","base_model":"llava:34b-v1.6-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q6_K":{"mode":"chat","base_model":"llava:34b-v1.6-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:34b-v1.6-q8_0":{"mode":"chat","base_model":"llava:34b-v1.6-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-fp16":{"mode":"chat","base_model":"llava:7b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q2_K":{"mode":"chat","base_model":"llava:7b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q3_K_L":{"mode":"chat","base_model":"llava:7b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q3_K_M":{"mode":"chat","base_model":"llava:7b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q3_K_S":{"mode":"chat","base_model":"llava:7b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q4_0":{"mode":"chat","base_model":"llava:7b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q4_1":{"mode":"chat","base_model":"llava:7b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q4_K_M":{"mode":"chat","base_model":"llava:7b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q4_K_S":{"mode":"chat","base_model":"llava:7b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q5_0":{"mode":"chat","base_model":"llava:7b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q5_1":{"mode":"chat","base_model":"llava:7b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q5_K_M":{"mode":"chat","base_model":"llava:7b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q5_K_S":{"mode":"chat","base_model":"llava:7b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q6_K":{"mode":"chat","base_model":"llava:7b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.5-q8_0":{"mode":"chat","base_model":"llava:7b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6":{"mode":"chat","base_model":"llava:7b-v1.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-fp16":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q2_K":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q3_K_L":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q3_K_M":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q3_K_S":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q4_0":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q4_1":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q4_K_M":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q4_K_S":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q5_0":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q5_1":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q5_K_M":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q5_K_S":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q6_K":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-mistral-q8_0":{"mode":"chat","base_model":"llava:7b-v1.6-mistral-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-fp16":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q2_K":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q3_K_L":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q3_K_M":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q3_K_S":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q4_0":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q4_1":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q4_K_M":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q4_K_S":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q5_0":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q5_1":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q5_K_M":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q5_K_S":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q6_K":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:7b-v1.6-vicuna-q8_0":{"mode":"chat","base_model":"llava:7b-v1.6-vicuna-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava:v1.6":{"mode":"chat","base_model":"llava:v1.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b":{"mode":"chat","base_model":"gemma2:2b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b":{"mode":"chat","base_model":"gemma2:27b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-fp16":{"mode":"chat","base_model":"gemma2:27b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q2_K":{"mode":"chat","base_model":"gemma2:27b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q3_K_L":{"mode":"chat","base_model":"gemma2:27b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q3_K_M":{"mode":"chat","base_model":"gemma2:27b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q3_K_S":{"mode":"chat","base_model":"gemma2:27b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q4_0":{"mode":"chat","base_model":"gemma2:27b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q4_1":{"mode":"chat","base_model":"gemma2:27b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q4_K_M":{"mode":"chat","base_model":"gemma2:27b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q4_K_S":{"mode":"chat","base_model":"gemma2:27b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q5_0":{"mode":"chat","base_model":"gemma2:27b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q5_1":{"mode":"chat","base_model":"gemma2:27b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q5_K_M":{"mode":"chat","base_model":"gemma2:27b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q5_K_S":{"mode":"chat","base_model":"gemma2:27b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q6_K":{"mode":"chat","base_model":"gemma2:27b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-instruct-q8_0":{"mode":"chat","base_model":"gemma2:27b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-fp16":{"mode":"chat","base_model":"gemma2:27b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q2_K":{"mode":"chat","base_model":"gemma2:27b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q3_K_L":{"mode":"chat","base_model":"gemma2:27b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q3_K_M":{"mode":"chat","base_model":"gemma2:27b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q3_K_S":{"mode":"chat","base_model":"gemma2:27b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q4_0":{"mode":"chat","base_model":"gemma2:27b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q4_1":{"mode":"chat","base_model":"gemma2:27b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q4_K_M":{"mode":"chat","base_model":"gemma2:27b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q4_K_S":{"mode":"chat","base_model":"gemma2:27b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q5_0":{"mode":"chat","base_model":"gemma2:27b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q5_1":{"mode":"chat","base_model":"gemma2:27b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q5_K_M":{"mode":"chat","base_model":"gemma2:27b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q5_K_S":{"mode":"chat","base_model":"gemma2:27b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q6_K":{"mode":"chat","base_model":"gemma2:27b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:27b-text-q8_0":{"mode":"chat","base_model":"gemma2:27b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-fp16":{"mode":"chat","base_model":"gemma2:2b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q2_K":{"mode":"chat","base_model":"gemma2:2b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q3_K_L":{"mode":"chat","base_model":"gemma2:2b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q3_K_M":{"mode":"chat","base_model":"gemma2:2b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q3_K_S":{"mode":"chat","base_model":"gemma2:2b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q4_0":{"mode":"chat","base_model":"gemma2:2b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q4_1":{"mode":"chat","base_model":"gemma2:2b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q4_K_M":{"mode":"chat","base_model":"gemma2:2b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q4_K_S":{"mode":"chat","base_model":"gemma2:2b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q5_0":{"mode":"chat","base_model":"gemma2:2b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q5_1":{"mode":"chat","base_model":"gemma2:2b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q5_K_M":{"mode":"chat","base_model":"gemma2:2b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q5_K_S":{"mode":"chat","base_model":"gemma2:2b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q6_K":{"mode":"chat","base_model":"gemma2:2b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-instruct-q8_0":{"mode":"chat","base_model":"gemma2:2b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-fp16":{"mode":"chat","base_model":"gemma2:2b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q2_K":{"mode":"chat","base_model":"gemma2:2b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q3_K_L":{"mode":"chat","base_model":"gemma2:2b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q3_K_M":{"mode":"chat","base_model":"gemma2:2b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q3_K_S":{"mode":"chat","base_model":"gemma2:2b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q4_0":{"mode":"chat","base_model":"gemma2:2b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q4_1":{"mode":"chat","base_model":"gemma2:2b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q4_K_M":{"mode":"chat","base_model":"gemma2:2b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q4_K_S":{"mode":"chat","base_model":"gemma2:2b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q5_0":{"mode":"chat","base_model":"gemma2:2b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q5_1":{"mode":"chat","base_model":"gemma2:2b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q5_K_M":{"mode":"chat","base_model":"gemma2:2b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q5_K_S":{"mode":"chat","base_model":"gemma2:2b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q6_K":{"mode":"chat","base_model":"gemma2:2b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:2b-text-q8_0":{"mode":"chat","base_model":"gemma2:2b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-fp16":{"mode":"chat","base_model":"gemma2:9b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q2_K":{"mode":"chat","base_model":"gemma2:9b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q3_K_L":{"mode":"chat","base_model":"gemma2:9b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q3_K_M":{"mode":"chat","base_model":"gemma2:9b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q3_K_S":{"mode":"chat","base_model":"gemma2:9b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q4_0":{"mode":"chat","base_model":"gemma2:9b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q4_1":{"mode":"chat","base_model":"gemma2:9b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q4_K_M":{"mode":"chat","base_model":"gemma2:9b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q4_K_S":{"mode":"chat","base_model":"gemma2:9b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q5_0":{"mode":"chat","base_model":"gemma2:9b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q5_1":{"mode":"chat","base_model":"gemma2:9b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q5_K_M":{"mode":"chat","base_model":"gemma2:9b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q5_K_S":{"mode":"chat","base_model":"gemma2:9b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q6_K":{"mode":"chat","base_model":"gemma2:9b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-instruct-q8_0":{"mode":"chat","base_model":"gemma2:9b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-fp16":{"mode":"chat","base_model":"gemma2:9b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q2_K":{"mode":"chat","base_model":"gemma2:9b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q3_K_L":{"mode":"chat","base_model":"gemma2:9b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q3_K_M":{"mode":"chat","base_model":"gemma2:9b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q3_K_S":{"mode":"chat","base_model":"gemma2:9b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q4_0":{"mode":"chat","base_model":"gemma2:9b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q4_1":{"mode":"chat","base_model":"gemma2:9b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q4_K_M":{"mode":"chat","base_model":"gemma2:9b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q4_K_S":{"mode":"chat","base_model":"gemma2:9b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q5_0":{"mode":"chat","base_model":"gemma2:9b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q5_1":{"mode":"chat","base_model":"gemma2:9b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q5_K_M":{"mode":"chat","base_model":"gemma2:9b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q5_K_S":{"mode":"chat","base_model":"gemma2:9b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q6_K":{"mode":"chat","base_model":"gemma2:9b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"gemma2:9b-text-q8_0":{"mode":"chat","base_model":"gemma2:9b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-fp16":{"mode":"chat","base_model":"llama2:13b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q2_K":{"mode":"chat","base_model":"llama2:13b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q3_K_L":{"mode":"chat","base_model":"llama2:13b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q3_K_M":{"mode":"chat","base_model":"llama2:13b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q3_K_S":{"mode":"chat","base_model":"llama2:13b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q4_0":{"mode":"chat","base_model":"llama2:13b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q4_1":{"mode":"chat","base_model":"llama2:13b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q4_K_M":{"mode":"chat","base_model":"llama2:13b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q4_K_S":{"mode":"chat","base_model":"llama2:13b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q5_0":{"mode":"chat","base_model":"llama2:13b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q5_1":{"mode":"chat","base_model":"llama2:13b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q5_K_M":{"mode":"chat","base_model":"llama2:13b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q5_K_S":{"mode":"chat","base_model":"llama2:13b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q6_K":{"mode":"chat","base_model":"llama2:13b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-chat-q8_0":{"mode":"chat","base_model":"llama2:13b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text":{"mode":"chat","base_model":"llama2:13b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-fp16":{"mode":"chat","base_model":"llama2:13b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q2_K":{"mode":"chat","base_model":"llama2:13b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q3_K_L":{"mode":"chat","base_model":"llama2:13b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q3_K_M":{"mode":"chat","base_model":"llama2:13b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q3_K_S":{"mode":"chat","base_model":"llama2:13b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q4_0":{"mode":"chat","base_model":"llama2:13b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q4_1":{"mode":"chat","base_model":"llama2:13b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q4_K_M":{"mode":"chat","base_model":"llama2:13b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q4_K_S":{"mode":"chat","base_model":"llama2:13b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q5_0":{"mode":"chat","base_model":"llama2:13b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q5_1":{"mode":"chat","base_model":"llama2:13b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q5_K_M":{"mode":"chat","base_model":"llama2:13b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q5_K_S":{"mode":"chat","base_model":"llama2:13b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q6_K":{"mode":"chat","base_model":"llama2:13b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:13b-text-q8_0":{"mode":"chat","base_model":"llama2:13b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-fp16":{"mode":"chat","base_model":"llama2:70b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q2_K":{"mode":"chat","base_model":"llama2:70b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q3_K_L":{"mode":"chat","base_model":"llama2:70b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q3_K_M":{"mode":"chat","base_model":"llama2:70b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q3_K_S":{"mode":"chat","base_model":"llama2:70b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q4_0":{"mode":"chat","base_model":"llama2:70b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q4_1":{"mode":"chat","base_model":"llama2:70b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q4_K_M":{"mode":"chat","base_model":"llama2:70b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q4_K_S":{"mode":"chat","base_model":"llama2:70b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q5_0":{"mode":"chat","base_model":"llama2:70b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q5_1":{"mode":"chat","base_model":"llama2:70b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q5_K_M":{"mode":"chat","base_model":"llama2:70b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q5_K_S":{"mode":"chat","base_model":"llama2:70b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q6_K":{"mode":"chat","base_model":"llama2:70b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-chat-q8_0":{"mode":"chat","base_model":"llama2:70b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text":{"mode":"chat","base_model":"llama2:70b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-fp16":{"mode":"chat","base_model":"llama2:70b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q2_K":{"mode":"chat","base_model":"llama2:70b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q3_K_L":{"mode":"chat","base_model":"llama2:70b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q3_K_M":{"mode":"chat","base_model":"llama2:70b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q3_K_S":{"mode":"chat","base_model":"llama2:70b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q4_0":{"mode":"chat","base_model":"llama2:70b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q4_1":{"mode":"chat","base_model":"llama2:70b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q4_K_M":{"mode":"chat","base_model":"llama2:70b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q4_K_S":{"mode":"chat","base_model":"llama2:70b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q5_0":{"mode":"chat","base_model":"llama2:70b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q5_1":{"mode":"chat","base_model":"llama2:70b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q5_K_M":{"mode":"chat","base_model":"llama2:70b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q5_K_S":{"mode":"chat","base_model":"llama2:70b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q6_K":{"mode":"chat","base_model":"llama2:70b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:70b-text-q8_0":{"mode":"chat","base_model":"llama2:70b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-fp16":{"mode":"chat","base_model":"llama2:7b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q2_K":{"mode":"chat","base_model":"llama2:7b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q3_K_L":{"mode":"chat","base_model":"llama2:7b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q3_K_M":{"mode":"chat","base_model":"llama2:7b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q3_K_S":{"mode":"chat","base_model":"llama2:7b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q4_0":{"mode":"chat","base_model":"llama2:7b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q4_1":{"mode":"chat","base_model":"llama2:7b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q4_K_M":{"mode":"chat","base_model":"llama2:7b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q4_K_S":{"mode":"chat","base_model":"llama2:7b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q5_0":{"mode":"chat","base_model":"llama2:7b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q5_1":{"mode":"chat","base_model":"llama2:7b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q5_K_M":{"mode":"chat","base_model":"llama2:7b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q5_K_S":{"mode":"chat","base_model":"llama2:7b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q6_K":{"mode":"chat","base_model":"llama2:7b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-chat-q8_0":{"mode":"chat","base_model":"llama2:7b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text":{"mode":"chat","base_model":"llama2:7b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-fp16":{"mode":"chat","base_model":"llama2:7b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q2_K":{"mode":"chat","base_model":"llama2:7b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q3_K_L":{"mode":"chat","base_model":"llama2:7b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q3_K_M":{"mode":"chat","base_model":"llama2:7b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q3_K_S":{"mode":"chat","base_model":"llama2:7b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q4_0":{"mode":"chat","base_model":"llama2:7b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q4_1":{"mode":"chat","base_model":"llama2:7b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q4_K_M":{"mode":"chat","base_model":"llama2:7b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q4_K_S":{"mode":"chat","base_model":"llama2:7b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q5_0":{"mode":"chat","base_model":"llama2:7b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q5_1":{"mode":"chat","base_model":"llama2:7b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q5_K_M":{"mode":"chat","base_model":"llama2:7b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q5_K_S":{"mode":"chat","base_model":"llama2:7b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q6_K":{"mode":"chat","base_model":"llama2:7b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:7b-text-q8_0":{"mode":"chat","base_model":"llama2:7b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:chat":{"mode":"chat","base_model":"llama2:chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2:text":{"mode":"chat","base_model":"llama2:text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b":{"mode":"chat","base_model":"phi3:3.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b":{"mode":"chat","base_model":"phi3:14b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-instruct":{"mode":"chat","base_model":"phi3:14b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-fp16":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q2_K":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q3_K_L":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q3_K_M":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q3_K_S":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q4_0":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q4_1":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q4_K_M":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q4_K_S":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q5_0":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q5_1":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q5_K_M":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q5_K_S":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q6_K":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-128k-instruct-q8_0":{"mode":"chat","base_model":"phi3:14b-medium-128k-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-fp16":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q2_K":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q3_K_L":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q3_K_M":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q3_K_S":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q4_0":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q4_1":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q4_K_M":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q4_K_S":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q5_0":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q5_1":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q5_K_M":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q5_K_S":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q6_K":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:14b-medium-4k-instruct-q8_0":{"mode":"chat","base_model":"phi3:14b-medium-4k-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-instruct":{"mode":"chat","base_model":"phi3:3.8b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-fp16":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q2_K":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q3_K_L":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q3_K_M":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q3_K_S":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q4_0":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q4_1":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q4_K_M":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q4_K_S":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q5_0":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q5_1":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q5_K_M":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q5_K_S":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q6_K":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-128k-instruct-q8_0":{"mode":"chat","base_model":"phi3:3.8b-mini-128k-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-fp16":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q2_K":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q3_K_L":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q3_K_M":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q3_K_S":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q4_0":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q4_1":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q4_K_M":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q4_K_S":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q5_0":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q5_1":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q5_K_M":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q5_K_S":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q6_K":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:3.8b-mini-4k-instruct-q8_0":{"mode":"chat","base_model":"phi3:3.8b-mini-4k-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3:instruct":{"mode":"chat","base_model":"phi3:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"phi3:medium":{"mode":"chat","base_model":"phi3:medium","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"phi3:medium-128k":{"mode":"chat","base_model":"phi3:medium-128k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"phi3:medium-4k":{"mode":"chat","base_model":"phi3:medium-4k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"phi3:mini":{"mode":"chat","base_model":"phi3:mini","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"phi3:mini-128k":{"mode":"chat","base_model":"phi3:mini-128k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"phi3:mini-4k":{"mode":"chat","base_model":"phi3:mini-4k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code":{"mode":"chat","base_model":"codellama:13b-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-fp16":{"mode":"chat","base_model":"codellama:13b-code-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q2_K":{"mode":"chat","base_model":"codellama:13b-code-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q3_K_L":{"mode":"chat","base_model":"codellama:13b-code-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q3_K_M":{"mode":"chat","base_model":"codellama:13b-code-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q3_K_S":{"mode":"chat","base_model":"codellama:13b-code-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q4_0":{"mode":"chat","base_model":"codellama:13b-code-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q4_1":{"mode":"chat","base_model":"codellama:13b-code-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q4_K_M":{"mode":"chat","base_model":"codellama:13b-code-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q4_K_S":{"mode":"chat","base_model":"codellama:13b-code-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q5_0":{"mode":"chat","base_model":"codellama:13b-code-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q5_1":{"mode":"chat","base_model":"codellama:13b-code-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q5_K_M":{"mode":"chat","base_model":"codellama:13b-code-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q5_K_S":{"mode":"chat","base_model":"codellama:13b-code-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q6_K":{"mode":"chat","base_model":"codellama:13b-code-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-code-q8_0":{"mode":"chat","base_model":"codellama:13b-code-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-fp16":{"mode":"chat","base_model":"codellama:13b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q2_K":{"mode":"chat","base_model":"codellama:13b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q3_K_L":{"mode":"chat","base_model":"codellama:13b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q3_K_M":{"mode":"chat","base_model":"codellama:13b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q3_K_S":{"mode":"chat","base_model":"codellama:13b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q4_0":{"mode":"chat","base_model":"codellama:13b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q4_1":{"mode":"chat","base_model":"codellama:13b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q4_K_M":{"mode":"chat","base_model":"codellama:13b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q4_K_S":{"mode":"chat","base_model":"codellama:13b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q5_0":{"mode":"chat","base_model":"codellama:13b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q5_1":{"mode":"chat","base_model":"codellama:13b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q5_K_M":{"mode":"chat","base_model":"codellama:13b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q5_K_S":{"mode":"chat","base_model":"codellama:13b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q6_K":{"mode":"chat","base_model":"codellama:13b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-instruct-q8_0":{"mode":"chat","base_model":"codellama:13b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-fp16":{"mode":"chat","base_model":"codellama:13b-python-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q2_K":{"mode":"chat","base_model":"codellama:13b-python-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q3_K_L":{"mode":"chat","base_model":"codellama:13b-python-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q3_K_M":{"mode":"chat","base_model":"codellama:13b-python-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q3_K_S":{"mode":"chat","base_model":"codellama:13b-python-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q4_0":{"mode":"chat","base_model":"codellama:13b-python-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q4_1":{"mode":"chat","base_model":"codellama:13b-python-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q4_K_M":{"mode":"chat","base_model":"codellama:13b-python-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q4_K_S":{"mode":"chat","base_model":"codellama:13b-python-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q5_0":{"mode":"chat","base_model":"codellama:13b-python-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q5_1":{"mode":"chat","base_model":"codellama:13b-python-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q5_K_M":{"mode":"chat","base_model":"codellama:13b-python-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q5_K_S":{"mode":"chat","base_model":"codellama:13b-python-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q6_K":{"mode":"chat","base_model":"codellama:13b-python-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:13b-python-q8_0":{"mode":"chat","base_model":"codellama:13b-python-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code":{"mode":"chat","base_model":"codellama:34b-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q2_K":{"mode":"chat","base_model":"codellama:34b-code-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q3_K_L":{"mode":"chat","base_model":"codellama:34b-code-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q3_K_M":{"mode":"chat","base_model":"codellama:34b-code-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q3_K_S":{"mode":"chat","base_model":"codellama:34b-code-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q4_0":{"mode":"chat","base_model":"codellama:34b-code-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q4_1":{"mode":"chat","base_model":"codellama:34b-code-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q4_K_M":{"mode":"chat","base_model":"codellama:34b-code-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q4_K_S":{"mode":"chat","base_model":"codellama:34b-code-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q5_0":{"mode":"chat","base_model":"codellama:34b-code-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q5_1":{"mode":"chat","base_model":"codellama:34b-code-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q5_K_M":{"mode":"chat","base_model":"codellama:34b-code-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q5_K_S":{"mode":"chat","base_model":"codellama:34b-code-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q6_K":{"mode":"chat","base_model":"codellama:34b-code-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-code-q8_0":{"mode":"chat","base_model":"codellama:34b-code-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-fp16":{"mode":"chat","base_model":"codellama:34b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q2_K":{"mode":"chat","base_model":"codellama:34b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q3_K_L":{"mode":"chat","base_model":"codellama:34b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q3_K_M":{"mode":"chat","base_model":"codellama:34b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q3_K_S":{"mode":"chat","base_model":"codellama:34b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q4_0":{"mode":"chat","base_model":"codellama:34b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q4_1":{"mode":"chat","base_model":"codellama:34b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q4_K_M":{"mode":"chat","base_model":"codellama:34b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q4_K_S":{"mode":"chat","base_model":"codellama:34b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q5_0":{"mode":"chat","base_model":"codellama:34b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q5_1":{"mode":"chat","base_model":"codellama:34b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q5_K_M":{"mode":"chat","base_model":"codellama:34b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q5_K_S":{"mode":"chat","base_model":"codellama:34b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q6_K":{"mode":"chat","base_model":"codellama:34b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-instruct-q8_0":{"mode":"chat","base_model":"codellama:34b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-fp16":{"mode":"chat","base_model":"codellama:34b-python-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q2_K":{"mode":"chat","base_model":"codellama:34b-python-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q3_K_L":{"mode":"chat","base_model":"codellama:34b-python-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q3_K_M":{"mode":"chat","base_model":"codellama:34b-python-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q3_K_S":{"mode":"chat","base_model":"codellama:34b-python-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q4_0":{"mode":"chat","base_model":"codellama:34b-python-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q4_1":{"mode":"chat","base_model":"codellama:34b-python-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q4_K_M":{"mode":"chat","base_model":"codellama:34b-python-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q4_K_S":{"mode":"chat","base_model":"codellama:34b-python-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q5_0":{"mode":"chat","base_model":"codellama:34b-python-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q5_1":{"mode":"chat","base_model":"codellama:34b-python-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q5_K_M":{"mode":"chat","base_model":"codellama:34b-python-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q5_K_S":{"mode":"chat","base_model":"codellama:34b-python-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q6_K":{"mode":"chat","base_model":"codellama:34b-python-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:34b-python-q8_0":{"mode":"chat","base_model":"codellama:34b-python-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code":{"mode":"chat","base_model":"codellama:70b-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-fp16":{"mode":"chat","base_model":"codellama:70b-code-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q2_K":{"mode":"chat","base_model":"codellama:70b-code-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q3_K_L":{"mode":"chat","base_model":"codellama:70b-code-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q3_K_M":{"mode":"chat","base_model":"codellama:70b-code-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q3_K_S":{"mode":"chat","base_model":"codellama:70b-code-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q4_0":{"mode":"chat","base_model":"codellama:70b-code-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q4_1":{"mode":"chat","base_model":"codellama:70b-code-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q4_K_M":{"mode":"chat","base_model":"codellama:70b-code-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q4_K_S":{"mode":"chat","base_model":"codellama:70b-code-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q5_0":{"mode":"chat","base_model":"codellama:70b-code-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q5_1":{"mode":"chat","base_model":"codellama:70b-code-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q5_K_M":{"mode":"chat","base_model":"codellama:70b-code-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q5_K_S":{"mode":"chat","base_model":"codellama:70b-code-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q6_K":{"mode":"chat","base_model":"codellama:70b-code-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-code-q8_0":{"mode":"chat","base_model":"codellama:70b-code-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-fp16":{"mode":"chat","base_model":"codellama:70b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q2_K":{"mode":"chat","base_model":"codellama:70b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q3_K_L":{"mode":"chat","base_model":"codellama:70b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q3_K_M":{"mode":"chat","base_model":"codellama:70b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q3_K_S":{"mode":"chat","base_model":"codellama:70b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q4_0":{"mode":"chat","base_model":"codellama:70b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q4_1":{"mode":"chat","base_model":"codellama:70b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q4_K_M":{"mode":"chat","base_model":"codellama:70b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q4_K_S":{"mode":"chat","base_model":"codellama:70b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q5_0":{"mode":"chat","base_model":"codellama:70b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q5_1":{"mode":"chat","base_model":"codellama:70b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q5_K_M":{"mode":"chat","base_model":"codellama:70b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q5_K_S":{"mode":"chat","base_model":"codellama:70b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q6_K":{"mode":"chat","base_model":"codellama:70b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-instruct-q8_0":{"mode":"chat","base_model":"codellama:70b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-fp16":{"mode":"chat","base_model":"codellama:70b-python-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q2_K":{"mode":"chat","base_model":"codellama:70b-python-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q3_K_L":{"mode":"chat","base_model":"codellama:70b-python-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q3_K_M":{"mode":"chat","base_model":"codellama:70b-python-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q3_K_S":{"mode":"chat","base_model":"codellama:70b-python-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q4_0":{"mode":"chat","base_model":"codellama:70b-python-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q4_1":{"mode":"chat","base_model":"codellama:70b-python-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q4_K_M":{"mode":"chat","base_model":"codellama:70b-python-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q4_K_S":{"mode":"chat","base_model":"codellama:70b-python-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q5_0":{"mode":"chat","base_model":"codellama:70b-python-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q5_1":{"mode":"chat","base_model":"codellama:70b-python-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q5_K_M":{"mode":"chat","base_model":"codellama:70b-python-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q5_K_S":{"mode":"chat","base_model":"codellama:70b-python-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q6_K":{"mode":"chat","base_model":"codellama:70b-python-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:70b-python-q8_0":{"mode":"chat","base_model":"codellama:70b-python-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code":{"mode":"chat","base_model":"codellama:7b-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-fp16":{"mode":"chat","base_model":"codellama:7b-code-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q2_K":{"mode":"chat","base_model":"codellama:7b-code-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q3_K_L":{"mode":"chat","base_model":"codellama:7b-code-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q3_K_M":{"mode":"chat","base_model":"codellama:7b-code-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q3_K_S":{"mode":"chat","base_model":"codellama:7b-code-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q4_0":{"mode":"chat","base_model":"codellama:7b-code-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q4_1":{"mode":"chat","base_model":"codellama:7b-code-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q4_K_M":{"mode":"chat","base_model":"codellama:7b-code-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q4_K_S":{"mode":"chat","base_model":"codellama:7b-code-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q5_0":{"mode":"chat","base_model":"codellama:7b-code-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q5_1":{"mode":"chat","base_model":"codellama:7b-code-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q5_K_M":{"mode":"chat","base_model":"codellama:7b-code-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q5_K_S":{"mode":"chat","base_model":"codellama:7b-code-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q6_K":{"mode":"chat","base_model":"codellama:7b-code-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-code-q8_0":{"mode":"chat","base_model":"codellama:7b-code-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-fp16":{"mode":"chat","base_model":"codellama:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q2_K":{"mode":"chat","base_model":"codellama:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q3_K_L":{"mode":"chat","base_model":"codellama:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q3_K_M":{"mode":"chat","base_model":"codellama:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q3_K_S":{"mode":"chat","base_model":"codellama:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q4_0":{"mode":"chat","base_model":"codellama:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q4_1":{"mode":"chat","base_model":"codellama:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q4_K_M":{"mode":"chat","base_model":"codellama:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q4_K_S":{"mode":"chat","base_model":"codellama:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q5_0":{"mode":"chat","base_model":"codellama:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q5_1":{"mode":"chat","base_model":"codellama:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q5_K_M":{"mode":"chat","base_model":"codellama:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q5_K_S":{"mode":"chat","base_model":"codellama:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q6_K":{"mode":"chat","base_model":"codellama:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-instruct-q8_0":{"mode":"chat","base_model":"codellama:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-fp16":{"mode":"chat","base_model":"codellama:7b-python-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q2_K":{"mode":"chat","base_model":"codellama:7b-python-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q3_K_L":{"mode":"chat","base_model":"codellama:7b-python-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q3_K_M":{"mode":"chat","base_model":"codellama:7b-python-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q3_K_S":{"mode":"chat","base_model":"codellama:7b-python-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q4_0":{"mode":"chat","base_model":"codellama:7b-python-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q4_1":{"mode":"chat","base_model":"codellama:7b-python-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q4_K_M":{"mode":"chat","base_model":"codellama:7b-python-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q4_K_S":{"mode":"chat","base_model":"codellama:7b-python-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q5_0":{"mode":"chat","base_model":"codellama:7b-python-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q5_1":{"mode":"chat","base_model":"codellama:7b-python-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q5_K_M":{"mode":"chat","base_model":"codellama:7b-python-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q5_K_S":{"mode":"chat","base_model":"codellama:7b-python-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q6_K":{"mode":"chat","base_model":"codellama:7b-python-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:7b-python-q8_0":{"mode":"chat","base_model":"codellama:7b-python-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:code":{"mode":"chat","base_model":"codellama:code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"codellama:instruct":{"mode":"chat","base_model":"codellama:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codellama:python":{"mode":"chat","base_model":"codellama:python","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mxbai-embed-large:335m":{"mode":"chat","base_model":"mxbai-embed-large:335m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mxbai-embed-large:335m-v1-fp16":{"mode":"chat","base_model":"mxbai-embed-large:335m-v1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mxbai-embed-large:v1":{"mode":"chat","base_model":"mxbai-embed-large:v1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3.2-vision:11b":{"mode":"chat","base_model":"llama3.2-vision:11b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"llama3.2-vision:90b":{"mode":"chat","base_model":"llama3.2-vision:90b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"llama3.2-vision:11b-instruct-fp16":{"mode":"chat","base_model":"llama3.2-vision:11b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"llama3.2-vision:11b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3.2-vision:11b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"llama3.2-vision:11b-instruct-q8_0":{"mode":"chat","base_model":"llama3.2-vision:11b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"llama3.2-vision:90b-instruct-fp16":{"mode":"chat","base_model":"llama3.2-vision:90b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"llama3.2-vision:90b-instruct-q4_K_M":{"mode":"chat","base_model":"llama3.2-vision:90b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"llama3.2-vision:90b-instruct-q8_0":{"mode":"chat","base_model":"llama3.2-vision:90b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b":{"mode":"chat","base_model":"tinyllama:1.1b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat":{"mode":"chat","base_model":"tinyllama:1.1b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-fp16":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q2_K":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q3_K_L":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q3_K_M":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q3_K_S":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q4_0":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q4_1":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q4_K_M":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q4_K_S":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q5_0":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q5_1":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q5_K_M":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q5_K_S":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q6_K":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v0.6-q8_0":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v0.6-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-fp16":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q2_K":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q3_K_L":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q3_K_M":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q3_K_S":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q4_0":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q4_1":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q4_K_M":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q4_K_S":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q5_0":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q5_1":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q5_K_M":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q5_K_S":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q6_K":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:1.1b-chat-v1-q8_0":{"mode":"chat","base_model":"tinyllama:1.1b-chat-v1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:chat":{"mode":"chat","base_model":"tinyllama:chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:v0.6":{"mode":"chat","base_model":"tinyllama:v0.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinyllama:v1":{"mode":"chat","base_model":"tinyllama:v1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b":{"mode":"chat","base_model":"mistral-nemo:12b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-fp16":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q2_K":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q3_K_L":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q3_K_M":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q3_K_S":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q4_0":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q4_1":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q4_K_M":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q4_K_S":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q5_0":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q5_1":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q5_K_M":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q5_K_S":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q6_K":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-nemo:12b-instruct-2407-q8_0":{"mode":"chat","base_model":"mistral-nemo:12b-instruct-2407-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-fp16":{"mode":"chat","base_model":"starcoder2:15b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct":{"mode":"chat","base_model":"starcoder2:15b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-q4_0":{"mode":"chat","base_model":"starcoder2:15b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-fp16":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q2_K":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q3_K_L":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q3_K_M":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q3_K_S":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q4_0":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q4_1":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q4_K_M":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q4_K_S":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q5_0":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q5_1":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q5_K_M":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q5_K_S":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q6_K":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-instruct-v0.1-q8_0":{"mode":"chat","base_model":"starcoder2:15b-instruct-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q2_K":{"mode":"chat","base_model":"starcoder2:15b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q3_K_L":{"mode":"chat","base_model":"starcoder2:15b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q3_K_M":{"mode":"chat","base_model":"starcoder2:15b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q3_K_S":{"mode":"chat","base_model":"starcoder2:15b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q4_0":{"mode":"chat","base_model":"starcoder2:15b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q4_1":{"mode":"chat","base_model":"starcoder2:15b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q4_K_M":{"mode":"chat","base_model":"starcoder2:15b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q4_K_S":{"mode":"chat","base_model":"starcoder2:15b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q5_0":{"mode":"chat","base_model":"starcoder2:15b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q5_1":{"mode":"chat","base_model":"starcoder2:15b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q5_K_M":{"mode":"chat","base_model":"starcoder2:15b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q5_K_S":{"mode":"chat","base_model":"starcoder2:15b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q6_K":{"mode":"chat","base_model":"starcoder2:15b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:15b-q8_0":{"mode":"chat","base_model":"starcoder2:15b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-fp16":{"mode":"chat","base_model":"starcoder2:3b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q2_K":{"mode":"chat","base_model":"starcoder2:3b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q3_K_L":{"mode":"chat","base_model":"starcoder2:3b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q3_K_M":{"mode":"chat","base_model":"starcoder2:3b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q3_K_S":{"mode":"chat","base_model":"starcoder2:3b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q4_0":{"mode":"chat","base_model":"starcoder2:3b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q4_1":{"mode":"chat","base_model":"starcoder2:3b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q4_K_M":{"mode":"chat","base_model":"starcoder2:3b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q4_K_S":{"mode":"chat","base_model":"starcoder2:3b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q5_0":{"mode":"chat","base_model":"starcoder2:3b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q5_1":{"mode":"chat","base_model":"starcoder2:3b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q5_K_M":{"mode":"chat","base_model":"starcoder2:3b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q5_K_S":{"mode":"chat","base_model":"starcoder2:3b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q6_K":{"mode":"chat","base_model":"starcoder2:3b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:3b-q8_0":{"mode":"chat","base_model":"starcoder2:3b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-fp16":{"mode":"chat","base_model":"starcoder2:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q2_K":{"mode":"chat","base_model":"starcoder2:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q3_K_L":{"mode":"chat","base_model":"starcoder2:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q3_K_M":{"mode":"chat","base_model":"starcoder2:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q3_K_S":{"mode":"chat","base_model":"starcoder2:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q4_0":{"mode":"chat","base_model":"starcoder2:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q4_1":{"mode":"chat","base_model":"starcoder2:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q4_K_M":{"mode":"chat","base_model":"starcoder2:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q4_K_S":{"mode":"chat","base_model":"starcoder2:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q5_0":{"mode":"chat","base_model":"starcoder2:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q5_1":{"mode":"chat","base_model":"starcoder2:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q5_K_M":{"mode":"chat","base_model":"starcoder2:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q5_K_S":{"mode":"chat","base_model":"starcoder2:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q6_K":{"mode":"chat","base_model":"starcoder2:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:7b-q8_0":{"mode":"chat","base_model":"starcoder2:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder2:instruct":{"mode":"chat","base_model":"starcoder2:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b":{"mode":"chat","base_model":"deepseek-coder-v2:16b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b":{"mode":"chat","base_model":"deepseek-coder-v2:236b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-fp16":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q2_K":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q3_K_L":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q3_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q3_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q4_0":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q4_1":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q4_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q4_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q5_0":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q5_1":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q5_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q5_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q6_K":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-base-q8_0":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-fp16":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q2_K":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q3_K_L":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q3_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q3_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q4_0":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q4_1":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q4_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q4_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q5_0":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q5_1":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q5_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q5_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q6_K":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:16b-lite-instruct-q8_0":{"mode":"chat","base_model":"deepseek-coder-v2:16b-lite-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-fp16":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q2_K":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q3_K_L":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q3_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q3_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q4_0":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q4_1":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q4_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q4_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q5_0":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q5_1":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q5_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q5_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q6_K":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-base-q8_0":{"mode":"chat","base_model":"deepseek-coder-v2:236b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-fp16":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q2_K":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q3_K_L":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q3_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q3_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q4_0":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q4_1":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q4_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q4_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q5_0":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q5_1":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q5_K_M":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q5_K_S":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q6_K":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:236b-instruct-q8_0":{"mode":"chat","base_model":"deepseek-coder-v2:236b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder-v2:lite":{"mode":"chat","base_model":"deepseek-coder-v2:lite","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:22m":{"mode":"chat","base_model":"snowflake-arctic-embed:22m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:33m":{"mode":"chat","base_model":"snowflake-arctic-embed:33m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:110m":{"mode":"chat","base_model":"snowflake-arctic-embed:110m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:137m":{"mode":"chat","base_model":"snowflake-arctic-embed:137m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:335m":{"mode":"chat","base_model":"snowflake-arctic-embed:335m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:110m-m-fp16":{"mode":"chat","base_model":"snowflake-arctic-embed:110m-m-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:137m-m-long-fp16":{"mode":"chat","base_model":"snowflake-arctic-embed:137m-m-long-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:22m-xs-fp16":{"mode":"chat","base_model":"snowflake-arctic-embed:22m-xs-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:335m-l-fp16":{"mode":"chat","base_model":"snowflake-arctic-embed:335m-l-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:33m-s-fp16":{"mode":"chat","base_model":"snowflake-arctic-embed:33m-s-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:l":{"mode":"chat","base_model":"snowflake-arctic-embed:l","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:m":{"mode":"chat","base_model":"snowflake-arctic-embed:m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:m-long":{"mode":"chat","base_model":"snowflake-arctic-embed:m-long","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:s":{"mode":"chat","base_model":"snowflake-arctic-embed:s","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed:xs":{"mode":"chat","base_model":"snowflake-arctic-embed:xs","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v3:671b":{"mode":"chat","base_model":"deepseek-v3:671b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v3:671b-fp16":{"mode":"chat","base_model":"deepseek-v3:671b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v3:671b-q4_K_M":{"mode":"chat","base_model":"deepseek-v3:671b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v3:671b-q8_0":{"mode":"chat","base_model":"deepseek-v3:671b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b":{"mode":"chat","base_model":"llama2-uncensored:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b":{"mode":"chat","base_model":"llama2-uncensored:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat":{"mode":"chat","base_model":"llama2-uncensored:70b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q2_K":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q3_K_L":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q3_K_M":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q3_K_S":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q4_0":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q4_1":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q4_K_M":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q4_K_S":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q5_0":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q5_1":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q5_K_M":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q5_K_S":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q6_K":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:70b-chat-q8_0":{"mode":"chat","base_model":"llama2-uncensored:70b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat":{"mode":"chat","base_model":"llama2-uncensored:7b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-fp16":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q2_K":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q3_K_L":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q3_K_M":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q3_K_S":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q4_0":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q4_1":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q4_K_M":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q4_K_S":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q5_0":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q5_1":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q5_K_M":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q5_K_S":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q6_K":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-uncensored:7b-chat-q8_0":{"mode":"chat","base_model":"llama2-uncensored:7b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b":{"mode":"chat","base_model":"deepseek-coder:1.3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b":{"mode":"chat","base_model":"deepseek-coder:33b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base":{"mode":"chat","base_model":"deepseek-coder:1.3b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-fp16":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q2_K":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q3_K_L":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q3_K_M":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q3_K_S":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q4_0":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q4_1":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q4_K_M":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q4_K_S":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q5_0":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q5_1":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q5_K_M":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q5_K_S":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q6_K":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-base-q8_0":{"mode":"chat","base_model":"deepseek-coder:1.3b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-fp16":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q2_K":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q3_K_L":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q3_K_M":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q3_K_S":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q4_0":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q4_1":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q4_K_M":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q4_K_S":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q5_0":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q5_1":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q5_K_M":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q5_K_S":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q6_K":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:1.3b-instruct-q8_0":{"mode":"chat","base_model":"deepseek-coder:1.3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base":{"mode":"chat","base_model":"deepseek-coder:33b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-fp16":{"mode":"chat","base_model":"deepseek-coder:33b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q2_K":{"mode":"chat","base_model":"deepseek-coder:33b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q3_K_L":{"mode":"chat","base_model":"deepseek-coder:33b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q3_K_M":{"mode":"chat","base_model":"deepseek-coder:33b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q3_K_S":{"mode":"chat","base_model":"deepseek-coder:33b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q4_0":{"mode":"chat","base_model":"deepseek-coder:33b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q4_1":{"mode":"chat","base_model":"deepseek-coder:33b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q4_K_M":{"mode":"chat","base_model":"deepseek-coder:33b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q4_K_S":{"mode":"chat","base_model":"deepseek-coder:33b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q5_0":{"mode":"chat","base_model":"deepseek-coder:33b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q5_1":{"mode":"chat","base_model":"deepseek-coder:33b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q5_K_M":{"mode":"chat","base_model":"deepseek-coder:33b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q5_K_S":{"mode":"chat","base_model":"deepseek-coder:33b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q6_K":{"mode":"chat","base_model":"deepseek-coder:33b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-base-q8_0":{"mode":"chat","base_model":"deepseek-coder:33b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-fp16":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q2_K":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q3_K_L":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q3_K_M":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q3_K_S":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q4_0":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q4_1":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q4_K_M":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q4_K_S":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q5_0":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q5_1":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q5_K_M":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q5_K_S":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q6_K":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:33b-instruct-q8_0":{"mode":"chat","base_model":"deepseek-coder:33b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base":{"mode":"chat","base_model":"deepseek-coder:6.7b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-fp16":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q2_K":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q3_K_L":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q3_K_M":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q3_K_S":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q4_0":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q4_1":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q4_K_M":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q4_K_S":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q5_0":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q5_1":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q5_K_M":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q5_K_S":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q6_K":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-base-q8_0":{"mode":"chat","base_model":"deepseek-coder:6.7b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-fp16":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q2_K":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q3_K_L":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q3_K_M":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q3_K_S":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q4_0":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q4_1":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q4_K_M":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q4_K_S":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q5_0":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q5_1":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q5_K_M":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q5_K_S":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q6_K":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:6.7b-instruct-q8_0":{"mode":"chat","base_model":"deepseek-coder:6.7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:base":{"mode":"chat","base_model":"deepseek-coder:base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-coder:instruct":{"mode":"chat","base_model":"deepseek-coder:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-fp16":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q2_K":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q3_K_L":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q3_K_M":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q3_K_S":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q4_0":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q4_1":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q4_K_M":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q4_K_S":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q5_0":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q5_1":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q5_K_M":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q5_K_S":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q6_K":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-instruct-v0.1-q8_0":{"mode":"chat","base_model":"mixtral:8x22b-instruct-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text":{"mode":"chat","base_model":"mixtral:8x22b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-fp16":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q2_K":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q3_K_L":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q3_K_M":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q3_K_S":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q4_0":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q4_1":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q4_K_M":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q4_K_S":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q5_0":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q5_1":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q5_K_M":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q5_K_S":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q6_K":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x22b-text-v0.1-q8_0":{"mode":"chat","base_model":"mixtral:8x22b-text-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-fp16":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q2_K":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q3_K_L":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q3_K_M":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q3_K_S":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q4_0":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q4_1":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q4_K_M":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q4_K_S":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q5_0":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q5_1":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q5_K_M":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q5_K_S":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q6_K":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-instruct-v0.1-q8_0":{"mode":"chat","base_model":"mixtral:8x7b-instruct-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text":{"mode":"chat","base_model":"mixtral:8x7b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-fp16":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q2_K":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q3_K_L":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q3_K_M":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q3_K_S":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q4_0":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q4_1":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q4_K_M":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q4_K_S":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q5_0":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q5_1":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q5_K_M":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q5_K_S":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q6_K":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:8x7b-text-v0.1-q8_0":{"mode":"chat","base_model":"mixtral:8x7b-text-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:instruct":{"mode":"chat","base_model":"mixtral:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:text":{"mode":"chat","base_model":"mixtral:text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:v0.1":{"mode":"chat","base_model":"mixtral:v0.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mixtral:v0.1-instruct":{"mode":"chat","base_model":"mixtral:v0.1-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b":{"mode":"chat","base_model":"dolphin-mixtral:8x7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b":{"mode":"chat","base_model":"dolphin-mixtral:8x22b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-fp16":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q2_K":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q3_K_L":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q3_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q3_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q4_0":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q4_1":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q4_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q4_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q5_0":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q5_1":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q5_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q5_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q6_K":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x22b-v2.9-q8_0":{"mode":"chat","base_model":"dolphin-mixtral:8x22b-v2.9-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-fp16":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q2_K":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q3_K_L":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q3_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q3_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q4_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q4_1":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q4_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q4_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q5_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q5_1":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q5_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q5_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q6_K":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.5-q8_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-fp16":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q2_K":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q3_K_L":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q3_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q3_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q4_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q4_1":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q4_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q4_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q5_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q5_1":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q5_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q5_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q6_K":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.6-q8_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.6-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-fp16":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q2_K":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q3_K_L":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q3_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q3_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q4_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q4_1":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q4_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q4_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q5_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q5_1":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q5_K_M":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q5_K_S":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q6_K":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:8x7b-v2.7-q8_0":{"mode":"chat","base_model":"dolphin-mixtral:8x7b-v2.7-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:v2.5":{"mode":"chat","base_model":"dolphin-mixtral:v2.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:v2.6":{"mode":"chat","base_model":"dolphin-mixtral:v2.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mixtral:v2.7":{"mode":"chat","base_model":"dolphin-mixtral:v2.7","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code":{"mode":"chat","base_model":"codegemma:2b-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-fp16":{"mode":"chat","base_model":"codegemma:2b-code-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q2_K":{"mode":"chat","base_model":"codegemma:2b-code-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q3_K_L":{"mode":"chat","base_model":"codegemma:2b-code-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q3_K_M":{"mode":"chat","base_model":"codegemma:2b-code-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q3_K_S":{"mode":"chat","base_model":"codegemma:2b-code-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q4_0":{"mode":"chat","base_model":"codegemma:2b-code-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q4_1":{"mode":"chat","base_model":"codegemma:2b-code-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q4_K_M":{"mode":"chat","base_model":"codegemma:2b-code-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q4_K_S":{"mode":"chat","base_model":"codegemma:2b-code-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q5_0":{"mode":"chat","base_model":"codegemma:2b-code-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q5_1":{"mode":"chat","base_model":"codegemma:2b-code-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q5_K_M":{"mode":"chat","base_model":"codegemma:2b-code-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q5_K_S":{"mode":"chat","base_model":"codegemma:2b-code-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q6_K":{"mode":"chat","base_model":"codegemma:2b-code-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-q8_0":{"mode":"chat","base_model":"codegemma:2b-code-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-fp16":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q2_K":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q3_K_L":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q3_K_M":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q3_K_S":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q4_0":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q4_1":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q4_K_M":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q4_K_S":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q5_0":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q5_1":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q5_K_M":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q5_K_S":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q6_K":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-code-v1.1-q8_0":{"mode":"chat","base_model":"codegemma:2b-code-v1.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:2b-v1.1":{"mode":"chat","base_model":"codegemma:2b-v1.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code":{"mode":"chat","base_model":"codegemma:7b-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-fp16":{"mode":"chat","base_model":"codegemma:7b-code-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q2_K":{"mode":"chat","base_model":"codegemma:7b-code-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q3_K_L":{"mode":"chat","base_model":"codegemma:7b-code-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q3_K_M":{"mode":"chat","base_model":"codegemma:7b-code-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q3_K_S":{"mode":"chat","base_model":"codegemma:7b-code-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q4_0":{"mode":"chat","base_model":"codegemma:7b-code-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q4_1":{"mode":"chat","base_model":"codegemma:7b-code-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q4_K_M":{"mode":"chat","base_model":"codegemma:7b-code-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q4_K_S":{"mode":"chat","base_model":"codegemma:7b-code-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q5_0":{"mode":"chat","base_model":"codegemma:7b-code-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q5_1":{"mode":"chat","base_model":"codegemma:7b-code-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q5_K_M":{"mode":"chat","base_model":"codegemma:7b-code-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q5_K_S":{"mode":"chat","base_model":"codegemma:7b-code-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q6_K":{"mode":"chat","base_model":"codegemma:7b-code-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-code-q8_0":{"mode":"chat","base_model":"codegemma:7b-code-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct":{"mode":"chat","base_model":"codegemma:7b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-fp16":{"mode":"chat","base_model":"codegemma:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q2_K":{"mode":"chat","base_model":"codegemma:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q3_K_L":{"mode":"chat","base_model":"codegemma:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q3_K_M":{"mode":"chat","base_model":"codegemma:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q3_K_S":{"mode":"chat","base_model":"codegemma:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q4_0":{"mode":"chat","base_model":"codegemma:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q4_1":{"mode":"chat","base_model":"codegemma:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q4_K_M":{"mode":"chat","base_model":"codegemma:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q4_K_S":{"mode":"chat","base_model":"codegemma:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q5_0":{"mode":"chat","base_model":"codegemma:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q5_1":{"mode":"chat","base_model":"codegemma:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q5_K_M":{"mode":"chat","base_model":"codegemma:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q5_K_S":{"mode":"chat","base_model":"codegemma:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q6_K":{"mode":"chat","base_model":"codegemma:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-q8_0":{"mode":"chat","base_model":"codegemma:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-fp16":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q2_K":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q3_K_L":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q3_K_M":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q3_K_S":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q4_0":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q4_1":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q4_K_M":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q4_K_S":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q5_0":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q5_1":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q5_K_M":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q5_K_S":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q6_K":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-instruct-v1.1-q8_0":{"mode":"chat","base_model":"codegemma:7b-instruct-v1.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:7b-v1.1":{"mode":"chat","base_model":"codegemma:7b-v1.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegemma:code":{"mode":"chat","base_model":"codegemma:code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"codegemma:instruct":{"mode":"chat","base_model":"codegemma:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openthinker:32b":{"mode":"chat","base_model":"openthinker:32b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openthinker:32b-fp16":{"mode":"chat","base_model":"openthinker:32b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openthinker:32b-q4_K_M":{"mode":"chat","base_model":"openthinker:32b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openthinker:32b-q8_0":{"mode":"chat","base_model":"openthinker:32b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openthinker:7b-fp16":{"mode":"chat","base_model":"openthinker:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openthinker:7b-q4_K_M":{"mode":"chat","base_model":"openthinker:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openthinker:7b-q8_0":{"mode":"chat","base_model":"openthinker:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b":{"mode":"chat","base_model":"phi:2.7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-fp16":{"mode":"chat","base_model":"phi:2.7b-chat-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q2_K":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q3_K_L":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q3_K_M":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q3_K_S":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q4_0":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q4_1":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q4_K_M":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q4_K_S":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q5_0":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q5_1":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q5_K_M":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q5_K_S":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q6_K":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:2.7b-chat-v2-q8_0":{"mode":"chat","base_model":"phi:2.7b-chat-v2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi:chat":{"mode":"chat","base_model":"phi:chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bge-m3:567m":{"mode":"chat","base_model":"bge-m3:567m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bge-m3:567m-fp16":{"mode":"chat","base_model":"bge-m3:567m-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b":{"mode":"chat","base_model":"minicpm-v:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-fp16":{"mode":"chat","base_model":"minicpm-v:8b-2.6-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q2_K":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q3_K_L":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q3_K_M":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q3_K_S":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q4_0":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q4_1":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q4_K_M":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q4_K_S":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q5_0":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q5_1":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q5_K_M":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q5_K_S":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q6_K":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"minicpm-v:8b-2.6-q8_0":{"mode":"chat","base_model":"minicpm-v:8b-2.6-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava-llama3:8b":{"mode":"chat","base_model":"llava-llama3:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava-llama3:8b-v1.1-fp16":{"mode":"chat","base_model":"llava-llama3:8b-v1.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava-llama3:8b-v1.1-q4_0":{"mode":"chat","base_model":"llava-llama3:8b-v1.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b":{"mode":"chat","base_model":"wizardlm2:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-fp16":{"mode":"chat","base_model":"wizardlm2:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q2_K":{"mode":"chat","base_model":"wizardlm2:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q3_K_L":{"mode":"chat","base_model":"wizardlm2:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q3_K_M":{"mode":"chat","base_model":"wizardlm2:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q3_K_S":{"mode":"chat","base_model":"wizardlm2:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q4_0":{"mode":"chat","base_model":"wizardlm2:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q4_1":{"mode":"chat","base_model":"wizardlm2:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q4_K_M":{"mode":"chat","base_model":"wizardlm2:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q4_K_S":{"mode":"chat","base_model":"wizardlm2:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q5_0":{"mode":"chat","base_model":"wizardlm2:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q5_1":{"mode":"chat","base_model":"wizardlm2:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q5_K_M":{"mode":"chat","base_model":"wizardlm2:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q5_K_S":{"mode":"chat","base_model":"wizardlm2:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q6_K":{"mode":"chat","base_model":"wizardlm2:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:7b-q8_0":{"mode":"chat","base_model":"wizardlm2:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:8x22b-fp16":{"mode":"chat","base_model":"wizardlm2:8x22b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:8x22b-q2_K":{"mode":"chat","base_model":"wizardlm2:8x22b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:8x22b-q4_0":{"mode":"chat","base_model":"wizardlm2:8x22b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm2:8x22b-q8_0":{"mode":"chat","base_model":"wizardlm2:8x22b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b":{"mode":"chat","base_model":"dolphin-mistral:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2":{"mode":"chat","base_model":"dolphin-mistral:7b-v2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-fp16":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q2_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q3_K_L":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q3_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q3_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q4_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q4_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q4_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q4_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q5_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q5_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q5_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q5_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q6_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2-q8_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-fp16":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q2_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q3_K_L":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q3_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q3_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q4_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q4_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q4_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q4_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q5_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q5_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q5_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q5_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q6_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.1-q8_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-fp16":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q2_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q3_K_L":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q3_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q3_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q4_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q4_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q4_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q4_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q5_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q5_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q5_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q5_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q6_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2-q8_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-fp16":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q2_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q3_K_L":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q3_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q3_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q4_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q4_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q4_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q4_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q5_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q5_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q5_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q5_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q6_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.2.1-q8_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.2.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-fp16":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q2_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q3_K_L":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q3_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q3_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q4_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q4_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q4_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q4_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q5_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q5_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q5_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q5_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q6_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-dpo-laser-q8_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-dpo-laser-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-fp16":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q2_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q3_K_L":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q3_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q3_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q4_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q4_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q4_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q4_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q5_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q5_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q5_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q5_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q6_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.6-q8_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.6-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-fp16":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q2_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q3_K_L":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q3_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q3_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q4_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q4_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q4_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q4_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q5_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q5_1":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q5_K_M":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q5_K_S":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q6_K":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:7b-v2.8-q8_0":{"mode":"chat","base_model":"dolphin-mistral:7b-v2.8-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:v2":{"mode":"chat","base_model":"dolphin-mistral:v2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:v2.1":{"mode":"chat","base_model":"dolphin-mistral:v2.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:v2.2":{"mode":"chat","base_model":"dolphin-mistral:v2.2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:v2.2.1":{"mode":"chat","base_model":"dolphin-mistral:v2.2.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:v2.6":{"mode":"chat","base_model":"dolphin-mistral:v2.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-mistral:v2.8":{"mode":"chat","base_model":"dolphin-mistral:v2.8","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:22m":{"mode":"chat","base_model":"all-minilm:22m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:33m":{"mode":"chat","base_model":"all-minilm:33m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:22m-l6-v2-fp16":{"mode":"chat","base_model":"all-minilm:22m-l6-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:33m-l12-v2-fp16":{"mode":"chat","base_model":"all-minilm:33m-l12-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:l12":{"mode":"chat","base_model":"all-minilm:l12","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:l12-v2":{"mode":"chat","base_model":"all-minilm:l12-v2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:l6":{"mode":"chat","base_model":"all-minilm:l6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:l6-v2":{"mode":"chat","base_model":"all-minilm:l6-v2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"all-minilm:v2":{"mode":"chat","base_model":"all-minilm:v2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m":{"mode":"chat","base_model":"smollm2:135m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m":{"mode":"chat","base_model":"smollm2:360m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b":{"mode":"chat","base_model":"smollm2:1.7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-fp16":{"mode":"chat","base_model":"smollm2:1.7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q2_K":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q3_K_L":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q3_K_M":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q3_K_S":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q4_0":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q4_1":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q4_K_M":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q4_K_S":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q5_0":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q5_1":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q5_K_M":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q5_K_S":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q6_K":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:1.7b-instruct-q8_0":{"mode":"chat","base_model":"smollm2:1.7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-fp16":{"mode":"chat","base_model":"smollm2:135m-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q2_K":{"mode":"chat","base_model":"smollm2:135m-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q3_K_L":{"mode":"chat","base_model":"smollm2:135m-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q3_K_M":{"mode":"chat","base_model":"smollm2:135m-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q3_K_S":{"mode":"chat","base_model":"smollm2:135m-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q4_0":{"mode":"chat","base_model":"smollm2:135m-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q4_1":{"mode":"chat","base_model":"smollm2:135m-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q4_K_M":{"mode":"chat","base_model":"smollm2:135m-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q4_K_S":{"mode":"chat","base_model":"smollm2:135m-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q5_0":{"mode":"chat","base_model":"smollm2:135m-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q5_1":{"mode":"chat","base_model":"smollm2:135m-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q5_K_M":{"mode":"chat","base_model":"smollm2:135m-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q5_K_S":{"mode":"chat","base_model":"smollm2:135m-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q6_K":{"mode":"chat","base_model":"smollm2:135m-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:135m-instruct-q8_0":{"mode":"chat","base_model":"smollm2:135m-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-fp16":{"mode":"chat","base_model":"smollm2:360m-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q2_K":{"mode":"chat","base_model":"smollm2:360m-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q3_K_L":{"mode":"chat","base_model":"smollm2:360m-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q3_K_M":{"mode":"chat","base_model":"smollm2:360m-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q3_K_S":{"mode":"chat","base_model":"smollm2:360m-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q4_0":{"mode":"chat","base_model":"smollm2:360m-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q4_1":{"mode":"chat","base_model":"smollm2:360m-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q4_K_M":{"mode":"chat","base_model":"smollm2:360m-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q4_K_S":{"mode":"chat","base_model":"smollm2:360m-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q5_0":{"mode":"chat","base_model":"smollm2:360m-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q5_1":{"mode":"chat","base_model":"smollm2:360m-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q5_K_M":{"mode":"chat","base_model":"smollm2:360m-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q5_K_S":{"mode":"chat","base_model":"smollm2:360m-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q6_K":{"mode":"chat","base_model":"smollm2:360m-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm2:360m-instruct-q8_0":{"mode":"chat","base_model":"smollm2:360m-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b":{"mode":"chat","base_model":"dolphin-llama3:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b":{"mode":"chat","base_model":"dolphin-llama3:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-fp16":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q2_K":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q3_K_L":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q3_K_M":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q3_K_S":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q4_0":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q4_1":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q4_K_M":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q4_K_S":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q5_0":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q5_1":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q5_K_M":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q5_K_S":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q6_K":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:70b-v2.9-q8_0":{"mode":"chat","base_model":"dolphin-llama3:70b-v2.9-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k":{"mode":"chat","base_model":"dolphin-llama3:8b-256k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-fp16":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q2_K":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q3_K_L":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q3_K_M":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q3_K_S":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q4_0":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q4_1":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q4_K_M":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q4_K_S":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q5_0":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q5_1":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q5_K_M":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q5_K_S":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q6_K":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-256k-v2.9-q8_0":{"mode":"chat","base_model":"dolphin-llama3:8b-256k-v2.9-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-fp16":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q2_K":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q3_K_L":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q3_K_M":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q3_K_S":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q4_0":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q4_1":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q4_K_M":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q4_K_S":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q5_0":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q5_1":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q5_K_M":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q5_K_S":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q6_K":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:8b-v2.9-q8_0":{"mode":"chat","base_model":"dolphin-llama3:8b-v2.9-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-llama3:v2.9":{"mode":"chat","base_model":"dolphin-llama3:v2.9","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b":{"mode":"chat","base_model":"command-r:35b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-fp16":{"mode":"chat","base_model":"command-r:35b-08-2024-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q2_K":{"mode":"chat","base_model":"command-r:35b-08-2024-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q3_K_L":{"mode":"chat","base_model":"command-r:35b-08-2024-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q3_K_M":{"mode":"chat","base_model":"command-r:35b-08-2024-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q3_K_S":{"mode":"chat","base_model":"command-r:35b-08-2024-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q4_0":{"mode":"chat","base_model":"command-r:35b-08-2024-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q4_1":{"mode":"chat","base_model":"command-r:35b-08-2024-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q4_K_M":{"mode":"chat","base_model":"command-r:35b-08-2024-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q4_K_S":{"mode":"chat","base_model":"command-r:35b-08-2024-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q5_0":{"mode":"chat","base_model":"command-r:35b-08-2024-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q5_1":{"mode":"chat","base_model":"command-r:35b-08-2024-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q5_K_M":{"mode":"chat","base_model":"command-r:35b-08-2024-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q5_K_S":{"mode":"chat","base_model":"command-r:35b-08-2024-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q6_K":{"mode":"chat","base_model":"command-r:35b-08-2024-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-08-2024-q8_0":{"mode":"chat","base_model":"command-r:35b-08-2024-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-fp16":{"mode":"chat","base_model":"command-r:35b-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q2_K":{"mode":"chat","base_model":"command-r:35b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q3_K_L":{"mode":"chat","base_model":"command-r:35b-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q3_K_M":{"mode":"chat","base_model":"command-r:35b-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q3_K_S":{"mode":"chat","base_model":"command-r:35b-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q4_0":{"mode":"chat","base_model":"command-r:35b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q4_1":{"mode":"chat","base_model":"command-r:35b-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q4_K_M":{"mode":"chat","base_model":"command-r:35b-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q4_K_S":{"mode":"chat","base_model":"command-r:35b-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q5_1":{"mode":"chat","base_model":"command-r:35b-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q5_K_M":{"mode":"chat","base_model":"command-r:35b-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q5_K_S":{"mode":"chat","base_model":"command-r:35b-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q6_K":{"mode":"chat","base_model":"command-r:35b-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:35b-v0.1-q8_0":{"mode":"chat","base_model":"command-r:35b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r:v0.1":{"mode":"chat","base_model":"command-r:v0.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:3b":{"mode":"chat","base_model":"orca-mini:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b":{"mode":"chat","base_model":"orca-mini:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b":{"mode":"chat","base_model":"orca-mini:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b":{"mode":"chat","base_model":"orca-mini:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-fp16":{"mode":"chat","base_model":"orca-mini:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q2_K":{"mode":"chat","base_model":"orca-mini:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q3_K_L":{"mode":"chat","base_model":"orca-mini:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q3_K_M":{"mode":"chat","base_model":"orca-mini:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q3_K_S":{"mode":"chat","base_model":"orca-mini:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q4_0":{"mode":"chat","base_model":"orca-mini:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q4_1":{"mode":"chat","base_model":"orca-mini:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q4_K_M":{"mode":"chat","base_model":"orca-mini:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q4_K_S":{"mode":"chat","base_model":"orca-mini:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q5_0":{"mode":"chat","base_model":"orca-mini:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q5_1":{"mode":"chat","base_model":"orca-mini:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q5_K_M":{"mode":"chat","base_model":"orca-mini:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q5_K_S":{"mode":"chat","base_model":"orca-mini:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q6_K":{"mode":"chat","base_model":"orca-mini:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-q8_0":{"mode":"chat","base_model":"orca-mini:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-fp16":{"mode":"chat","base_model":"orca-mini:13b-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q2_K":{"mode":"chat","base_model":"orca-mini:13b-v2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q3_K_L":{"mode":"chat","base_model":"orca-mini:13b-v2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q3_K_M":{"mode":"chat","base_model":"orca-mini:13b-v2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q3_K_S":{"mode":"chat","base_model":"orca-mini:13b-v2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q4_0":{"mode":"chat","base_model":"orca-mini:13b-v2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q4_1":{"mode":"chat","base_model":"orca-mini:13b-v2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q4_K_M":{"mode":"chat","base_model":"orca-mini:13b-v2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q4_K_S":{"mode":"chat","base_model":"orca-mini:13b-v2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q5_0":{"mode":"chat","base_model":"orca-mini:13b-v2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q5_1":{"mode":"chat","base_model":"orca-mini:13b-v2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q5_K_M":{"mode":"chat","base_model":"orca-mini:13b-v2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q5_K_S":{"mode":"chat","base_model":"orca-mini:13b-v2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q6_K":{"mode":"chat","base_model":"orca-mini:13b-v2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v2-q8_0":{"mode":"chat","base_model":"orca-mini:13b-v2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3":{"mode":"chat","base_model":"orca-mini:13b-v3","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-fp16":{"mode":"chat","base_model":"orca-mini:13b-v3-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q2_K":{"mode":"chat","base_model":"orca-mini:13b-v3-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q3_K_L":{"mode":"chat","base_model":"orca-mini:13b-v3-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q3_K_M":{"mode":"chat","base_model":"orca-mini:13b-v3-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q3_K_S":{"mode":"chat","base_model":"orca-mini:13b-v3-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q4_0":{"mode":"chat","base_model":"orca-mini:13b-v3-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q4_1":{"mode":"chat","base_model":"orca-mini:13b-v3-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q4_K_M":{"mode":"chat","base_model":"orca-mini:13b-v3-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q4_K_S":{"mode":"chat","base_model":"orca-mini:13b-v3-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q5_0":{"mode":"chat","base_model":"orca-mini:13b-v3-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q5_1":{"mode":"chat","base_model":"orca-mini:13b-v3-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q5_K_M":{"mode":"chat","base_model":"orca-mini:13b-v3-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q5_K_S":{"mode":"chat","base_model":"orca-mini:13b-v3-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q6_K":{"mode":"chat","base_model":"orca-mini:13b-v3-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:13b-v3-q8_0":{"mode":"chat","base_model":"orca-mini:13b-v3-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:3b-fp16":{"mode":"chat","base_model":"orca-mini:3b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:3b-q4_0":{"mode":"chat","base_model":"orca-mini:3b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:3b-q4_1":{"mode":"chat","base_model":"orca-mini:3b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:3b-q5_0":{"mode":"chat","base_model":"orca-mini:3b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:3b-q5_1":{"mode":"chat","base_model":"orca-mini:3b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:3b-q8_0":{"mode":"chat","base_model":"orca-mini:3b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3":{"mode":"chat","base_model":"orca-mini:70b-v3","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-fp16":{"mode":"chat","base_model":"orca-mini:70b-v3-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q2_K":{"mode":"chat","base_model":"orca-mini:70b-v3-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q3_K_L":{"mode":"chat","base_model":"orca-mini:70b-v3-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q3_K_M":{"mode":"chat","base_model":"orca-mini:70b-v3-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q3_K_S":{"mode":"chat","base_model":"orca-mini:70b-v3-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q4_0":{"mode":"chat","base_model":"orca-mini:70b-v3-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q4_1":{"mode":"chat","base_model":"orca-mini:70b-v3-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q4_K_M":{"mode":"chat","base_model":"orca-mini:70b-v3-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q4_K_S":{"mode":"chat","base_model":"orca-mini:70b-v3-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q5_0":{"mode":"chat","base_model":"orca-mini:70b-v3-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q5_1":{"mode":"chat","base_model":"orca-mini:70b-v3-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q5_K_M":{"mode":"chat","base_model":"orca-mini:70b-v3-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q5_K_S":{"mode":"chat","base_model":"orca-mini:70b-v3-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q6_K":{"mode":"chat","base_model":"orca-mini:70b-v3-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:70b-v3-q8_0":{"mode":"chat","base_model":"orca-mini:70b-v3-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-fp16":{"mode":"chat","base_model":"orca-mini:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q2_K":{"mode":"chat","base_model":"orca-mini:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q3_K_L":{"mode":"chat","base_model":"orca-mini:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q3_K_M":{"mode":"chat","base_model":"orca-mini:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q3_K_S":{"mode":"chat","base_model":"orca-mini:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q4_0":{"mode":"chat","base_model":"orca-mini:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q4_1":{"mode":"chat","base_model":"orca-mini:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q4_K_M":{"mode":"chat","base_model":"orca-mini:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q4_K_S":{"mode":"chat","base_model":"orca-mini:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q5_0":{"mode":"chat","base_model":"orca-mini:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q5_1":{"mode":"chat","base_model":"orca-mini:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q5_K_M":{"mode":"chat","base_model":"orca-mini:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q5_K_S":{"mode":"chat","base_model":"orca-mini:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q6_K":{"mode":"chat","base_model":"orca-mini:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-q8_0":{"mode":"chat","base_model":"orca-mini:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-fp16":{"mode":"chat","base_model":"orca-mini:7b-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q2_K":{"mode":"chat","base_model":"orca-mini:7b-v2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q3_K_L":{"mode":"chat","base_model":"orca-mini:7b-v2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q3_K_M":{"mode":"chat","base_model":"orca-mini:7b-v2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q3_K_S":{"mode":"chat","base_model":"orca-mini:7b-v2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q4_0":{"mode":"chat","base_model":"orca-mini:7b-v2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q4_1":{"mode":"chat","base_model":"orca-mini:7b-v2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q4_K_M":{"mode":"chat","base_model":"orca-mini:7b-v2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q4_K_S":{"mode":"chat","base_model":"orca-mini:7b-v2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q5_0":{"mode":"chat","base_model":"orca-mini:7b-v2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q5_1":{"mode":"chat","base_model":"orca-mini:7b-v2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q5_K_M":{"mode":"chat","base_model":"orca-mini:7b-v2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q5_K_S":{"mode":"chat","base_model":"orca-mini:7b-v2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q6_K":{"mode":"chat","base_model":"orca-mini:7b-v2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v2-q8_0":{"mode":"chat","base_model":"orca-mini:7b-v2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3":{"mode":"chat","base_model":"orca-mini:7b-v3","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-fp16":{"mode":"chat","base_model":"orca-mini:7b-v3-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q2_K":{"mode":"chat","base_model":"orca-mini:7b-v3-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q3_K_L":{"mode":"chat","base_model":"orca-mini:7b-v3-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q3_K_M":{"mode":"chat","base_model":"orca-mini:7b-v3-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q3_K_S":{"mode":"chat","base_model":"orca-mini:7b-v3-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q4_0":{"mode":"chat","base_model":"orca-mini:7b-v3-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q4_1":{"mode":"chat","base_model":"orca-mini:7b-v3-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q4_K_M":{"mode":"chat","base_model":"orca-mini:7b-v3-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q4_K_S":{"mode":"chat","base_model":"orca-mini:7b-v3-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q5_0":{"mode":"chat","base_model":"orca-mini:7b-v3-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q5_1":{"mode":"chat","base_model":"orca-mini:7b-v3-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q5_K_M":{"mode":"chat","base_model":"orca-mini:7b-v3-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q5_K_S":{"mode":"chat","base_model":"orca-mini:7b-v3-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q6_K":{"mode":"chat","base_model":"orca-mini:7b-v3-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca-mini:7b-v3-q8_0":{"mode":"chat","base_model":"orca-mini:7b-v3-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b":{"mode":"chat","base_model":"yi:9b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-fp16":{"mode":"chat","base_model":"yi:34b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q2_K":{"mode":"chat","base_model":"yi:34b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q3_K_L":{"mode":"chat","base_model":"yi:34b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q3_K_M":{"mode":"chat","base_model":"yi:34b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q3_K_S":{"mode":"chat","base_model":"yi:34b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q4_0":{"mode":"chat","base_model":"yi:34b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q4_1":{"mode":"chat","base_model":"yi:34b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q4_K_M":{"mode":"chat","base_model":"yi:34b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q4_K_S":{"mode":"chat","base_model":"yi:34b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q5_0":{"mode":"chat","base_model":"yi:34b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q5_1":{"mode":"chat","base_model":"yi:34b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q5_K_M":{"mode":"chat","base_model":"yi:34b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q5_K_S":{"mode":"chat","base_model":"yi:34b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q6_K":{"mode":"chat","base_model":"yi:34b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-q8_0":{"mode":"chat","base_model":"yi:34b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-fp16":{"mode":"chat","base_model":"yi:34b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q2_K":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q4_0":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q4_1":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q5_0":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q5_1":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q6_K":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-chat-v1.5-q8_0":{"mode":"chat","base_model":"yi:34b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q2_K":{"mode":"chat","base_model":"yi:34b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q3_K_L":{"mode":"chat","base_model":"yi:34b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q3_K_M":{"mode":"chat","base_model":"yi:34b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q3_K_S":{"mode":"chat","base_model":"yi:34b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q4_0":{"mode":"chat","base_model":"yi:34b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q4_1":{"mode":"chat","base_model":"yi:34b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q4_K_M":{"mode":"chat","base_model":"yi:34b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q4_K_S":{"mode":"chat","base_model":"yi:34b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q5_0":{"mode":"chat","base_model":"yi:34b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q5_1":{"mode":"chat","base_model":"yi:34b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q5_K_S":{"mode":"chat","base_model":"yi:34b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-q6_K":{"mode":"chat","base_model":"yi:34b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5":{"mode":"chat","base_model":"yi:34b-v1.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-fp16":{"mode":"chat","base_model":"yi:34b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q2_K":{"mode":"chat","base_model":"yi:34b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q3_K_L":{"mode":"chat","base_model":"yi:34b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q3_K_M":{"mode":"chat","base_model":"yi:34b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q3_K_S":{"mode":"chat","base_model":"yi:34b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q4_0":{"mode":"chat","base_model":"yi:34b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q4_1":{"mode":"chat","base_model":"yi:34b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q4_K_M":{"mode":"chat","base_model":"yi:34b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q4_K_S":{"mode":"chat","base_model":"yi:34b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q5_0":{"mode":"chat","base_model":"yi:34b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q5_1":{"mode":"chat","base_model":"yi:34b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q5_K_M":{"mode":"chat","base_model":"yi:34b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q5_K_S":{"mode":"chat","base_model":"yi:34b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q6_K":{"mode":"chat","base_model":"yi:34b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:34b-v1.5-q8_0":{"mode":"chat","base_model":"yi:34b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k":{"mode":"chat","base_model":"yi:6b-200k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-fp16":{"mode":"chat","base_model":"yi:6b-200k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q2_K":{"mode":"chat","base_model":"yi:6b-200k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q3_K_L":{"mode":"chat","base_model":"yi:6b-200k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q3_K_M":{"mode":"chat","base_model":"yi:6b-200k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q3_K_S":{"mode":"chat","base_model":"yi:6b-200k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q4_0":{"mode":"chat","base_model":"yi:6b-200k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q4_1":{"mode":"chat","base_model":"yi:6b-200k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q4_K_M":{"mode":"chat","base_model":"yi:6b-200k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q4_K_S":{"mode":"chat","base_model":"yi:6b-200k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q5_0":{"mode":"chat","base_model":"yi:6b-200k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q5_1":{"mode":"chat","base_model":"yi:6b-200k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q5_K_M":{"mode":"chat","base_model":"yi:6b-200k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q5_K_S":{"mode":"chat","base_model":"yi:6b-200k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q6_K":{"mode":"chat","base_model":"yi:6b-200k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-200k-q8_0":{"mode":"chat","base_model":"yi:6b-200k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat":{"mode":"chat","base_model":"yi:6b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-fp16":{"mode":"chat","base_model":"yi:6b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q2_K":{"mode":"chat","base_model":"yi:6b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q3_K_L":{"mode":"chat","base_model":"yi:6b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q3_K_M":{"mode":"chat","base_model":"yi:6b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q3_K_S":{"mode":"chat","base_model":"yi:6b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q4_0":{"mode":"chat","base_model":"yi:6b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q4_1":{"mode":"chat","base_model":"yi:6b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q4_K_M":{"mode":"chat","base_model":"yi:6b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q4_K_S":{"mode":"chat","base_model":"yi:6b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q5_0":{"mode":"chat","base_model":"yi:6b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q5_1":{"mode":"chat","base_model":"yi:6b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q5_K_M":{"mode":"chat","base_model":"yi:6b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q5_K_S":{"mode":"chat","base_model":"yi:6b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q6_K":{"mode":"chat","base_model":"yi:6b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-q8_0":{"mode":"chat","base_model":"yi:6b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-fp16":{"mode":"chat","base_model":"yi:6b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q2_K":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q4_0":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q4_1":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q5_0":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q5_1":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q6_K":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-chat-v1.5-q8_0":{"mode":"chat","base_model":"yi:6b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-fp16":{"mode":"chat","base_model":"yi:6b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q2_K":{"mode":"chat","base_model":"yi:6b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q3_K_L":{"mode":"chat","base_model":"yi:6b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q3_K_M":{"mode":"chat","base_model":"yi:6b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q3_K_S":{"mode":"chat","base_model":"yi:6b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q4_0":{"mode":"chat","base_model":"yi:6b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q4_1":{"mode":"chat","base_model":"yi:6b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q4_K_M":{"mode":"chat","base_model":"yi:6b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q4_K_S":{"mode":"chat","base_model":"yi:6b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q5_0":{"mode":"chat","base_model":"yi:6b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q5_1":{"mode":"chat","base_model":"yi:6b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q5_K_M":{"mode":"chat","base_model":"yi:6b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q5_K_S":{"mode":"chat","base_model":"yi:6b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q6_K":{"mode":"chat","base_model":"yi:6b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-q8_0":{"mode":"chat","base_model":"yi:6b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5":{"mode":"chat","base_model":"yi:6b-v1.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-fp16":{"mode":"chat","base_model":"yi:6b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q2_K":{"mode":"chat","base_model":"yi:6b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q3_K_L":{"mode":"chat","base_model":"yi:6b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q3_K_M":{"mode":"chat","base_model":"yi:6b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q3_K_S":{"mode":"chat","base_model":"yi:6b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q4_0":{"mode":"chat","base_model":"yi:6b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q4_1":{"mode":"chat","base_model":"yi:6b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q4_K_M":{"mode":"chat","base_model":"yi:6b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q4_K_S":{"mode":"chat","base_model":"yi:6b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q5_0":{"mode":"chat","base_model":"yi:6b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q5_1":{"mode":"chat","base_model":"yi:6b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q5_K_M":{"mode":"chat","base_model":"yi:6b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q5_K_S":{"mode":"chat","base_model":"yi:6b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q6_K":{"mode":"chat","base_model":"yi:6b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:6b-v1.5-q8_0":{"mode":"chat","base_model":"yi:6b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat":{"mode":"chat","base_model":"yi:9b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-fp16":{"mode":"chat","base_model":"yi:9b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q2_K":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q4_0":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q4_1":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q5_0":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q5_1":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q6_K":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-chat-v1.5-q8_0":{"mode":"chat","base_model":"yi:9b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5":{"mode":"chat","base_model":"yi:9b-v1.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-fp16":{"mode":"chat","base_model":"yi:9b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q2_K":{"mode":"chat","base_model":"yi:9b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q3_K_L":{"mode":"chat","base_model":"yi:9b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q3_K_M":{"mode":"chat","base_model":"yi:9b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q3_K_S":{"mode":"chat","base_model":"yi:9b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q4_0":{"mode":"chat","base_model":"yi:9b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q4_1":{"mode":"chat","base_model":"yi:9b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q4_K_M":{"mode":"chat","base_model":"yi:9b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q4_K_S":{"mode":"chat","base_model":"yi:9b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q5_0":{"mode":"chat","base_model":"yi:9b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q5_1":{"mode":"chat","base_model":"yi:9b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q5_K_M":{"mode":"chat","base_model":"yi:9b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q5_K_S":{"mode":"chat","base_model":"yi:9b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q6_K":{"mode":"chat","base_model":"yi:9b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:9b-v1.5-q8_0":{"mode":"chat","base_model":"yi:9b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi:v1.5":{"mode":"chat","base_model":"yi:v1.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b":{"mode":"chat","base_model":"hermes3:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-fp16":{"mode":"chat","base_model":"hermes3:3b-llama3.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q2_K":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q3_K_L":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q3_K_M":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q3_K_S":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q4_0":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q4_1":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q4_K_M":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q4_K_S":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q5_0":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q5_1":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q5_K_M":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q5_K_S":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q6_K":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:3b-llama3.2-q8_0":{"mode":"chat","base_model":"hermes3:3b-llama3.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-fp16":{"mode":"chat","base_model":"hermes3:405b-llama3.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q2_K":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q3_K_L":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q3_K_M":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q3_K_S":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q4_0":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q4_1":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q4_K_M":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q4_K_S":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q5_0":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q5_1":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q5_K_M":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q5_K_S":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q6_K":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:405b-llama3.1-q8_0":{"mode":"chat","base_model":"hermes3:405b-llama3.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-fp16":{"mode":"chat","base_model":"hermes3:70b-llama3.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q2_K":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q3_K_L":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q3_K_M":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q3_K_S":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q4_0":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q4_1":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q4_K_M":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q4_K_S":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q5_0":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q5_1":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q5_K_M":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q5_K_S":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q6_K":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:70b-llama3.1-q8_0":{"mode":"chat","base_model":"hermes3:70b-llama3.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-fp16":{"mode":"chat","base_model":"hermes3:8b-llama3.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q2_K":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q3_K_L":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q3_K_M":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q3_K_S":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q4_0":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q4_1":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q4_K_M":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q4_K_S":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q5_0":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q5_1":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q5_K_M":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q5_K_S":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q6_K":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"hermes3:8b-llama3.1-q8_0":{"mode":"chat","base_model":"hermes3:8b-llama3.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin3:8b-llama3.1-fp16":{"mode":"chat","base_model":"dolphin3:8b-llama3.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin3:8b-llama3.1-q4_K_M":{"mode":"chat","base_model":"dolphin3:8b-llama3.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin3:8b-llama3.1-q8_0":{"mode":"chat","base_model":"dolphin3:8b-llama3.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b":{"mode":"chat","base_model":"phi3.5:3.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-fp16":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q2_K":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q3_K_L":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q3_K_M":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q3_K_S":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q4_0":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q4_1":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q4_K_M":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q4_K_S":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q5_0":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q5_1":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q5_K_M":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q5_K_S":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q6_K":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi3.5:3.8b-mini-instruct-q8_0":{"mode":"chat","base_model":"phi3.5:3.8b-mini-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:141b":{"mode":"chat","base_model":"zephyr:141b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:141b-v0.1":{"mode":"chat","base_model":"zephyr:141b-v0.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:141b-v0.1-fp16":{"mode":"chat","base_model":"zephyr:141b-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:141b-v0.1-q2_K":{"mode":"chat","base_model":"zephyr:141b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:141b-v0.1-q4_0":{"mode":"chat","base_model":"zephyr:141b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:141b-v0.1-q8_0":{"mode":"chat","base_model":"zephyr:141b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha":{"mode":"chat","base_model":"zephyr:7b-alpha","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-fp16":{"mode":"chat","base_model":"zephyr:7b-alpha-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q2_K":{"mode":"chat","base_model":"zephyr:7b-alpha-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q3_K_L":{"mode":"chat","base_model":"zephyr:7b-alpha-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q3_K_M":{"mode":"chat","base_model":"zephyr:7b-alpha-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q3_K_S":{"mode":"chat","base_model":"zephyr:7b-alpha-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q4_0":{"mode":"chat","base_model":"zephyr:7b-alpha-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q4_1":{"mode":"chat","base_model":"zephyr:7b-alpha-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q4_K_M":{"mode":"chat","base_model":"zephyr:7b-alpha-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q4_K_S":{"mode":"chat","base_model":"zephyr:7b-alpha-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q5_0":{"mode":"chat","base_model":"zephyr:7b-alpha-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q5_1":{"mode":"chat","base_model":"zephyr:7b-alpha-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q5_K_M":{"mode":"chat","base_model":"zephyr:7b-alpha-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q5_K_S":{"mode":"chat","base_model":"zephyr:7b-alpha-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q6_K":{"mode":"chat","base_model":"zephyr:7b-alpha-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-alpha-q8_0":{"mode":"chat","base_model":"zephyr:7b-alpha-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta":{"mode":"chat","base_model":"zephyr:7b-beta","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-fp16":{"mode":"chat","base_model":"zephyr:7b-beta-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q2_K":{"mode":"chat","base_model":"zephyr:7b-beta-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q3_K_L":{"mode":"chat","base_model":"zephyr:7b-beta-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q3_K_M":{"mode":"chat","base_model":"zephyr:7b-beta-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q3_K_S":{"mode":"chat","base_model":"zephyr:7b-beta-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q4_0":{"mode":"chat","base_model":"zephyr:7b-beta-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q4_1":{"mode":"chat","base_model":"zephyr:7b-beta-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q4_K_M":{"mode":"chat","base_model":"zephyr:7b-beta-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q4_K_S":{"mode":"chat","base_model":"zephyr:7b-beta-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q5_0":{"mode":"chat","base_model":"zephyr:7b-beta-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q5_1":{"mode":"chat","base_model":"zephyr:7b-beta-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q5_K_M":{"mode":"chat","base_model":"zephyr:7b-beta-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q5_K_S":{"mode":"chat","base_model":"zephyr:7b-beta-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q6_K":{"mode":"chat","base_model":"zephyr:7b-beta-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"zephyr:7b-beta-q8_0":{"mode":"chat","base_model":"zephyr:7b-beta-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"olmo2:7b":{"mode":"chat","base_model":"olmo2:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"olmo2:13b":{"mode":"chat","base_model":"olmo2:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"olmo2:13b-1124-instruct-fp16":{"mode":"chat","base_model":"olmo2:13b-1124-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"olmo2:13b-1124-instruct-q4_K_M":{"mode":"chat","base_model":"olmo2:13b-1124-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"olmo2:13b-1124-instruct-q8_0":{"mode":"chat","base_model":"olmo2:13b-1124-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"olmo2:7b-1124-instruct-fp16":{"mode":"chat","base_model":"olmo2:7b-1124-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"olmo2:7b-1124-instruct-q4_K_M":{"mode":"chat","base_model":"olmo2:7b-1124-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"olmo2:7b-1124-instruct-q8_0":{"mode":"chat","base_model":"olmo2:7b-1124-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b":{"mode":"chat","base_model":"mistral-small:22b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:24b":{"mode":"chat","base_model":"mistral-small:24b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-fp16":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q2_K":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q3_K_L":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q3_K_M":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q3_K_S":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q4_0":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q4_1":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q4_K_M":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q4_K_S":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q5_0":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q5_1":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q5_K_M":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q5_K_S":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q6_K":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:22b-instruct-2409-q8_0":{"mode":"chat","base_model":"mistral-small:22b-instruct-2409-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:24b-instruct-2501-fp16":{"mode":"chat","base_model":"mistral-small:24b-instruct-2501-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:24b-instruct-2501-q4_K_M":{"mode":"chat","base_model":"mistral-small:24b-instruct-2501-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-small:24b-instruct-2501-q8_0":{"mode":"chat","base_model":"mistral-small:24b-instruct-2501-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b":{"mode":"chat","base_model":"codestral:22b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q2_K":{"mode":"chat","base_model":"codestral:22b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q3_K_L":{"mode":"chat","base_model":"codestral:22b-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q3_K_M":{"mode":"chat","base_model":"codestral:22b-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q3_K_S":{"mode":"chat","base_model":"codestral:22b-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q4_0":{"mode":"chat","base_model":"codestral:22b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q4_1":{"mode":"chat","base_model":"codestral:22b-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q4_K_M":{"mode":"chat","base_model":"codestral:22b-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q4_K_S":{"mode":"chat","base_model":"codestral:22b-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q5_0":{"mode":"chat","base_model":"codestral:22b-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q5_1":{"mode":"chat","base_model":"codestral:22b-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q5_K_M":{"mode":"chat","base_model":"codestral:22b-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q5_K_S":{"mode":"chat","base_model":"codestral:22b-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q6_K":{"mode":"chat","base_model":"codestral:22b-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:22b-v0.1-q8_0":{"mode":"chat","base_model":"codestral:22b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codestral:v0.1":{"mode":"chat","base_model":"codestral:v0.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b":{"mode":"chat","base_model":"granite-code:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b":{"mode":"chat","base_model":"granite-code:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b":{"mode":"chat","base_model":"granite-code:20b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b":{"mode":"chat","base_model":"granite-code:34b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base":{"mode":"chat","base_model":"granite-code:20b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-fp16":{"mode":"chat","base_model":"granite-code:20b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q2_K":{"mode":"chat","base_model":"granite-code:20b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q3_K_L":{"mode":"chat","base_model":"granite-code:20b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q3_K_M":{"mode":"chat","base_model":"granite-code:20b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q3_K_S":{"mode":"chat","base_model":"granite-code:20b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q4_0":{"mode":"chat","base_model":"granite-code:20b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q4_1":{"mode":"chat","base_model":"granite-code:20b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q4_K_M":{"mode":"chat","base_model":"granite-code:20b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q4_K_S":{"mode":"chat","base_model":"granite-code:20b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q5_0":{"mode":"chat","base_model":"granite-code:20b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q5_1":{"mode":"chat","base_model":"granite-code:20b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q5_K_M":{"mode":"chat","base_model":"granite-code:20b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q5_K_S":{"mode":"chat","base_model":"granite-code:20b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q6_K":{"mode":"chat","base_model":"granite-code:20b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-base-q8_0":{"mode":"chat","base_model":"granite-code:20b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct":{"mode":"chat","base_model":"granite-code:20b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-fp16":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q2_K":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q3_K_L":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q3_K_M":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q3_K_S":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q4_0":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q4_1":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q4_K_M":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q4_K_S":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q5_0":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q5_1":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q5_K_M":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q5_K_S":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q6_K":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-8k-q8_0":{"mode":"chat","base_model":"granite-code:20b-instruct-8k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q2_K":{"mode":"chat","base_model":"granite-code:20b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q3_K_L":{"mode":"chat","base_model":"granite-code:20b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q3_K_M":{"mode":"chat","base_model":"granite-code:20b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q3_K_S":{"mode":"chat","base_model":"granite-code:20b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q4_0":{"mode":"chat","base_model":"granite-code:20b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q4_1":{"mode":"chat","base_model":"granite-code:20b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q4_K_M":{"mode":"chat","base_model":"granite-code:20b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q4_K_S":{"mode":"chat","base_model":"granite-code:20b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q5_0":{"mode":"chat","base_model":"granite-code:20b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q5_1":{"mode":"chat","base_model":"granite-code:20b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q5_K_M":{"mode":"chat","base_model":"granite-code:20b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q5_K_S":{"mode":"chat","base_model":"granite-code:20b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q6_K":{"mode":"chat","base_model":"granite-code:20b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:20b-instruct-q8_0":{"mode":"chat","base_model":"granite-code:20b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base":{"mode":"chat","base_model":"granite-code:34b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q2_K":{"mode":"chat","base_model":"granite-code:34b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q3_K_L":{"mode":"chat","base_model":"granite-code:34b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q3_K_M":{"mode":"chat","base_model":"granite-code:34b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q3_K_S":{"mode":"chat","base_model":"granite-code:34b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q4_0":{"mode":"chat","base_model":"granite-code:34b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q4_1":{"mode":"chat","base_model":"granite-code:34b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q4_K_M":{"mode":"chat","base_model":"granite-code:34b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q4_K_S":{"mode":"chat","base_model":"granite-code:34b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q5_0":{"mode":"chat","base_model":"granite-code:34b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q5_1":{"mode":"chat","base_model":"granite-code:34b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q5_K_M":{"mode":"chat","base_model":"granite-code:34b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q5_K_S":{"mode":"chat","base_model":"granite-code:34b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q6_K":{"mode":"chat","base_model":"granite-code:34b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-base-q8_0":{"mode":"chat","base_model":"granite-code:34b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct":{"mode":"chat","base_model":"granite-code:34b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q2_K":{"mode":"chat","base_model":"granite-code:34b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q3_K_L":{"mode":"chat","base_model":"granite-code:34b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q3_K_M":{"mode":"chat","base_model":"granite-code:34b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q3_K_S":{"mode":"chat","base_model":"granite-code:34b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q4_0":{"mode":"chat","base_model":"granite-code:34b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q4_1":{"mode":"chat","base_model":"granite-code:34b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q4_K_M":{"mode":"chat","base_model":"granite-code:34b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q4_K_S":{"mode":"chat","base_model":"granite-code:34b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q5_0":{"mode":"chat","base_model":"granite-code:34b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q5_1":{"mode":"chat","base_model":"granite-code:34b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q5_K_M":{"mode":"chat","base_model":"granite-code:34b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q5_K_S":{"mode":"chat","base_model":"granite-code:34b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q6_K":{"mode":"chat","base_model":"granite-code:34b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:34b-instruct-q8_0":{"mode":"chat","base_model":"granite-code:34b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base":{"mode":"chat","base_model":"granite-code:3b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-fp16":{"mode":"chat","base_model":"granite-code:3b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q2_K":{"mode":"chat","base_model":"granite-code:3b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q3_K_L":{"mode":"chat","base_model":"granite-code:3b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q3_K_M":{"mode":"chat","base_model":"granite-code:3b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q3_K_S":{"mode":"chat","base_model":"granite-code:3b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q4_0":{"mode":"chat","base_model":"granite-code:3b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q4_1":{"mode":"chat","base_model":"granite-code:3b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q4_K_M":{"mode":"chat","base_model":"granite-code:3b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q4_K_S":{"mode":"chat","base_model":"granite-code:3b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q5_0":{"mode":"chat","base_model":"granite-code:3b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q5_1":{"mode":"chat","base_model":"granite-code:3b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q5_K_M":{"mode":"chat","base_model":"granite-code:3b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q5_K_S":{"mode":"chat","base_model":"granite-code:3b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q6_K":{"mode":"chat","base_model":"granite-code:3b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-base-q8_0":{"mode":"chat","base_model":"granite-code:3b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct":{"mode":"chat","base_model":"granite-code:3b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-fp16":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q2_K":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q3_K_L":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q3_K_M":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q3_K_S":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q4_0":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q4_1":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q4_K_M":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q4_K_S":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q5_0":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q5_1":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q5_K_M":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q5_K_S":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q6_K":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-128k-q8_0":{"mode":"chat","base_model":"granite-code:3b-instruct-128k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-fp16":{"mode":"chat","base_model":"granite-code:3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q2_K":{"mode":"chat","base_model":"granite-code:3b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q3_K_L":{"mode":"chat","base_model":"granite-code:3b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q3_K_M":{"mode":"chat","base_model":"granite-code:3b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q3_K_S":{"mode":"chat","base_model":"granite-code:3b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q4_0":{"mode":"chat","base_model":"granite-code:3b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q4_1":{"mode":"chat","base_model":"granite-code:3b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q4_K_M":{"mode":"chat","base_model":"granite-code:3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q4_K_S":{"mode":"chat","base_model":"granite-code:3b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q5_0":{"mode":"chat","base_model":"granite-code:3b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q5_1":{"mode":"chat","base_model":"granite-code:3b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q5_K_M":{"mode":"chat","base_model":"granite-code:3b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q5_K_S":{"mode":"chat","base_model":"granite-code:3b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q6_K":{"mode":"chat","base_model":"granite-code:3b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:3b-instruct-q8_0":{"mode":"chat","base_model":"granite-code:3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base":{"mode":"chat","base_model":"granite-code:8b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-fp16":{"mode":"chat","base_model":"granite-code:8b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q2_K":{"mode":"chat","base_model":"granite-code:8b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q3_K_L":{"mode":"chat","base_model":"granite-code:8b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q3_K_M":{"mode":"chat","base_model":"granite-code:8b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q3_K_S":{"mode":"chat","base_model":"granite-code:8b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q4_0":{"mode":"chat","base_model":"granite-code:8b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q4_1":{"mode":"chat","base_model":"granite-code:8b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q4_K_M":{"mode":"chat","base_model":"granite-code:8b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q4_K_S":{"mode":"chat","base_model":"granite-code:8b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q5_0":{"mode":"chat","base_model":"granite-code:8b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q5_1":{"mode":"chat","base_model":"granite-code:8b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q5_K_M":{"mode":"chat","base_model":"granite-code:8b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q5_K_S":{"mode":"chat","base_model":"granite-code:8b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q6_K":{"mode":"chat","base_model":"granite-code:8b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-base-q8_0":{"mode":"chat","base_model":"granite-code:8b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct":{"mode":"chat","base_model":"granite-code:8b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-128k-q4_0":{"mode":"chat","base_model":"granite-code:8b-instruct-128k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-128k-q4_1":{"mode":"chat","base_model":"granite-code:8b-instruct-128k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-fp16":{"mode":"chat","base_model":"granite-code:8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q2_K":{"mode":"chat","base_model":"granite-code:8b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q3_K_L":{"mode":"chat","base_model":"granite-code:8b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q3_K_M":{"mode":"chat","base_model":"granite-code:8b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q3_K_S":{"mode":"chat","base_model":"granite-code:8b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q4_0":{"mode":"chat","base_model":"granite-code:8b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q4_1":{"mode":"chat","base_model":"granite-code:8b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q4_K_M":{"mode":"chat","base_model":"granite-code:8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q4_K_S":{"mode":"chat","base_model":"granite-code:8b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q5_0":{"mode":"chat","base_model":"granite-code:8b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q5_1":{"mode":"chat","base_model":"granite-code:8b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q5_K_M":{"mode":"chat","base_model":"granite-code:8b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q5_K_S":{"mode":"chat","base_model":"granite-code:8b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q6_K":{"mode":"chat","base_model":"granite-code:8b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-code:8b-instruct-q8_0":{"mode":"chat","base_model":"granite-code:8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b":{"mode":"chat","base_model":"starcoder:1b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b":{"mode":"chat","base_model":"starcoder:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b":{"mode":"chat","base_model":"starcoder:15b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base":{"mode":"chat","base_model":"starcoder:15b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-fp16":{"mode":"chat","base_model":"starcoder:15b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q2_K":{"mode":"chat","base_model":"starcoder:15b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q3_K_L":{"mode":"chat","base_model":"starcoder:15b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q3_K_M":{"mode":"chat","base_model":"starcoder:15b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q3_K_S":{"mode":"chat","base_model":"starcoder:15b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q4_0":{"mode":"chat","base_model":"starcoder:15b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q4_1":{"mode":"chat","base_model":"starcoder:15b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q4_K_M":{"mode":"chat","base_model":"starcoder:15b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q4_K_S":{"mode":"chat","base_model":"starcoder:15b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q5_0":{"mode":"chat","base_model":"starcoder:15b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q5_1":{"mode":"chat","base_model":"starcoder:15b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q5_K_M":{"mode":"chat","base_model":"starcoder:15b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q5_K_S":{"mode":"chat","base_model":"starcoder:15b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q6_K":{"mode":"chat","base_model":"starcoder:15b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-base-q8_0":{"mode":"chat","base_model":"starcoder:15b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-fp16":{"mode":"chat","base_model":"starcoder:15b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus":{"mode":"chat","base_model":"starcoder:15b-plus","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-fp16":{"mode":"chat","base_model":"starcoder:15b-plus-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q2_K":{"mode":"chat","base_model":"starcoder:15b-plus-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q3_K_L":{"mode":"chat","base_model":"starcoder:15b-plus-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q3_K_M":{"mode":"chat","base_model":"starcoder:15b-plus-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q3_K_S":{"mode":"chat","base_model":"starcoder:15b-plus-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q4_0":{"mode":"chat","base_model":"starcoder:15b-plus-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q4_1":{"mode":"chat","base_model":"starcoder:15b-plus-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q4_K_M":{"mode":"chat","base_model":"starcoder:15b-plus-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q4_K_S":{"mode":"chat","base_model":"starcoder:15b-plus-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q5_0":{"mode":"chat","base_model":"starcoder:15b-plus-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q5_1":{"mode":"chat","base_model":"starcoder:15b-plus-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q5_K_M":{"mode":"chat","base_model":"starcoder:15b-plus-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q5_K_S":{"mode":"chat","base_model":"starcoder:15b-plus-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q6_K":{"mode":"chat","base_model":"starcoder:15b-plus-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-plus-q8_0":{"mode":"chat","base_model":"starcoder:15b-plus-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q2_K":{"mode":"chat","base_model":"starcoder:15b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q3_K_L":{"mode":"chat","base_model":"starcoder:15b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q3_K_M":{"mode":"chat","base_model":"starcoder:15b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q3_K_S":{"mode":"chat","base_model":"starcoder:15b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q4_0":{"mode":"chat","base_model":"starcoder:15b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q4_1":{"mode":"chat","base_model":"starcoder:15b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q4_K_M":{"mode":"chat","base_model":"starcoder:15b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q4_K_S":{"mode":"chat","base_model":"starcoder:15b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q5_0":{"mode":"chat","base_model":"starcoder:15b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q5_1":{"mode":"chat","base_model":"starcoder:15b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q5_K_M":{"mode":"chat","base_model":"starcoder:15b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q5_K_S":{"mode":"chat","base_model":"starcoder:15b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q6_K":{"mode":"chat","base_model":"starcoder:15b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:15b-q8_0":{"mode":"chat","base_model":"starcoder:15b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base":{"mode":"chat","base_model":"starcoder:1b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-fp16":{"mode":"chat","base_model":"starcoder:1b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q2_K":{"mode":"chat","base_model":"starcoder:1b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q3_K_L":{"mode":"chat","base_model":"starcoder:1b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q3_K_M":{"mode":"chat","base_model":"starcoder:1b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q3_K_S":{"mode":"chat","base_model":"starcoder:1b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q4_0":{"mode":"chat","base_model":"starcoder:1b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q4_1":{"mode":"chat","base_model":"starcoder:1b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q4_K_M":{"mode":"chat","base_model":"starcoder:1b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q4_K_S":{"mode":"chat","base_model":"starcoder:1b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q5_0":{"mode":"chat","base_model":"starcoder:1b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q5_1":{"mode":"chat","base_model":"starcoder:1b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q5_K_M":{"mode":"chat","base_model":"starcoder:1b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q5_K_S":{"mode":"chat","base_model":"starcoder:1b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q6_K":{"mode":"chat","base_model":"starcoder:1b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:1b-base-q8_0":{"mode":"chat","base_model":"starcoder:1b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base":{"mode":"chat","base_model":"starcoder:3b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-fp16":{"mode":"chat","base_model":"starcoder:3b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q2_K":{"mode":"chat","base_model":"starcoder:3b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q3_K_L":{"mode":"chat","base_model":"starcoder:3b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q3_K_M":{"mode":"chat","base_model":"starcoder:3b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q3_K_S":{"mode":"chat","base_model":"starcoder:3b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q4_0":{"mode":"chat","base_model":"starcoder:3b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q4_1":{"mode":"chat","base_model":"starcoder:3b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q4_K_M":{"mode":"chat","base_model":"starcoder:3b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q4_K_S":{"mode":"chat","base_model":"starcoder:3b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q5_0":{"mode":"chat","base_model":"starcoder:3b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q5_1":{"mode":"chat","base_model":"starcoder:3b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q5_K_M":{"mode":"chat","base_model":"starcoder:3b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q5_K_S":{"mode":"chat","base_model":"starcoder:3b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q6_K":{"mode":"chat","base_model":"starcoder:3b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:3b-base-q8_0":{"mode":"chat","base_model":"starcoder:3b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base":{"mode":"chat","base_model":"starcoder:7b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-fp16":{"mode":"chat","base_model":"starcoder:7b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q2_K":{"mode":"chat","base_model":"starcoder:7b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q3_K_L":{"mode":"chat","base_model":"starcoder:7b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q3_K_M":{"mode":"chat","base_model":"starcoder:7b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q3_K_S":{"mode":"chat","base_model":"starcoder:7b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q4_0":{"mode":"chat","base_model":"starcoder:7b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q4_1":{"mode":"chat","base_model":"starcoder:7b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q4_K_M":{"mode":"chat","base_model":"starcoder:7b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q4_K_S":{"mode":"chat","base_model":"starcoder:7b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q5_0":{"mode":"chat","base_model":"starcoder:7b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q5_1":{"mode":"chat","base_model":"starcoder:7b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q5_K_M":{"mode":"chat","base_model":"starcoder:7b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q5_K_S":{"mode":"chat","base_model":"starcoder:7b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q6_K":{"mode":"chat","base_model":"starcoder:7b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starcoder:7b-base-q8_0":{"mode":"chat","base_model":"starcoder:7b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m":{"mode":"chat","base_model":"smollm:135m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m":{"mode":"chat","base_model":"smollm:360m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b":{"mode":"chat","base_model":"smollm:1.7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-fp16":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q2_K":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q3_K_L":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q3_K_M":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q3_K_S":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q4_0":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q4_1":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q4_K_M":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q4_K_S":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q5_0":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q5_1":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q5_K_M":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q5_K_S":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q6_K":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-base-v0.2-q8_0":{"mode":"chat","base_model":"smollm:1.7b-base-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-fp16":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q2_K":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q3_K_L":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q3_K_M":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q3_K_S":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q4_0":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q4_1":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q4_K_M":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q4_K_S":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q5_0":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q5_1":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q5_K_M":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q5_K_S":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q6_K":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:1.7b-instruct-v0.2-q8_0":{"mode":"chat","base_model":"smollm:1.7b-instruct-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-fp16":{"mode":"chat","base_model":"smollm:135m-base-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q2_K":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q3_K_L":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q3_K_M":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q3_K_S":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q4_0":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q4_1":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q4_K_M":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q4_K_S":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q5_0":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q5_1":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q5_K_M":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q5_K_S":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q6_K":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-base-v0.2-q8_0":{"mode":"chat","base_model":"smollm:135m-base-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-fp16":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q2_K":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q3_K_L":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q3_K_M":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q3_K_S":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q4_0":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q4_1":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q4_K_M":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q4_K_S":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q5_0":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q5_1":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q5_K_M":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q5_K_S":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q6_K":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:135m-instruct-v0.2-q8_0":{"mode":"chat","base_model":"smollm:135m-instruct-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-fp16":{"mode":"chat","base_model":"smollm:360m-base-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q2_K":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q3_K_L":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q3_K_M":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q3_K_S":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q4_0":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q4_1":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q4_K_M":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q4_K_S":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q5_0":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q5_1":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q5_K_M":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q5_K_S":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q6_K":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-base-v0.2-q8_0":{"mode":"chat","base_model":"smollm:360m-base-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-fp16":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q2_K":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q3_K_L":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q3_K_M":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q3_K_S":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q4_0":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q4_1":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q4_K_M":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q4_K_S":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q5_0":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q5_1":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q5_K_M":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q5_K_S":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q6_K":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smollm:360m-instruct-v0.2-q8_0":{"mode":"chat","base_model":"smollm:360m-instruct-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-fp16":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q2_K":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q3_K_L":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q3_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q3_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q4_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q4_1":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q4_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q4_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q5_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q5_1":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q5_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q5_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q6_K":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:13b-q8_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-fp16":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q2_K":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q3_K_L":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q3_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q3_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q4_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q4_1":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q4_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q4_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q5_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q5_1":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q5_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q5_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q6_K":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:30b-q8_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:30b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-fp16":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q2_K":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q3_K_L":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q3_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q3_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q4_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q4_1":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q4_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q4_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q5_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q5_1":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q5_K_M":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q5_K_S":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q6_K":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna-uncensored:7b-q8_0":{"mode":"chat","base_model":"wizard-vicuna-uncensored:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b":{"mode":"chat","base_model":"vicuna:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b":{"mode":"chat","base_model":"vicuna:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b":{"mode":"chat","base_model":"vicuna:33b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-16k":{"mode":"chat","base_model":"vicuna:13b-16k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-fp16":{"mode":"chat","base_model":"vicuna:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q2_K":{"mode":"chat","base_model":"vicuna:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q3_K_L":{"mode":"chat","base_model":"vicuna:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q3_K_M":{"mode":"chat","base_model":"vicuna:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q3_K_S":{"mode":"chat","base_model":"vicuna:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q4_0":{"mode":"chat","base_model":"vicuna:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q4_1":{"mode":"chat","base_model":"vicuna:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q4_K_M":{"mode":"chat","base_model":"vicuna:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q4_K_S":{"mode":"chat","base_model":"vicuna:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q5_0":{"mode":"chat","base_model":"vicuna:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q5_1":{"mode":"chat","base_model":"vicuna:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q5_K_M":{"mode":"chat","base_model":"vicuna:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q5_K_S":{"mode":"chat","base_model":"vicuna:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q6_K":{"mode":"chat","base_model":"vicuna:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-q8_0":{"mode":"chat","base_model":"vicuna:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-fp16":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q2_K":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q3_K_L":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q3_K_M":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q3_K_S":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q4_0":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q4_1":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q4_K_M":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q4_K_S":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q5_0":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q5_1":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q5_K_M":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q5_K_S":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q6_K":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-16k-q8_0":{"mode":"chat","base_model":"vicuna:13b-v1.5-16k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-fp16":{"mode":"chat","base_model":"vicuna:13b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q2_K":{"mode":"chat","base_model":"vicuna:13b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q3_K_L":{"mode":"chat","base_model":"vicuna:13b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q3_K_M":{"mode":"chat","base_model":"vicuna:13b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q3_K_S":{"mode":"chat","base_model":"vicuna:13b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q4_0":{"mode":"chat","base_model":"vicuna:13b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q4_1":{"mode":"chat","base_model":"vicuna:13b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q4_K_M":{"mode":"chat","base_model":"vicuna:13b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q4_K_S":{"mode":"chat","base_model":"vicuna:13b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q5_0":{"mode":"chat","base_model":"vicuna:13b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q5_1":{"mode":"chat","base_model":"vicuna:13b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q5_K_M":{"mode":"chat","base_model":"vicuna:13b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q5_K_S":{"mode":"chat","base_model":"vicuna:13b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q6_K":{"mode":"chat","base_model":"vicuna:13b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:13b-v1.5-q8_0":{"mode":"chat","base_model":"vicuna:13b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-fp16":{"mode":"chat","base_model":"vicuna:33b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q2_K":{"mode":"chat","base_model":"vicuna:33b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q3_K_L":{"mode":"chat","base_model":"vicuna:33b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q3_K_M":{"mode":"chat","base_model":"vicuna:33b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q3_K_S":{"mode":"chat","base_model":"vicuna:33b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q4_0":{"mode":"chat","base_model":"vicuna:33b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q4_1":{"mode":"chat","base_model":"vicuna:33b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q4_K_M":{"mode":"chat","base_model":"vicuna:33b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q4_K_S":{"mode":"chat","base_model":"vicuna:33b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q5_0":{"mode":"chat","base_model":"vicuna:33b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q5_1":{"mode":"chat","base_model":"vicuna:33b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q5_K_M":{"mode":"chat","base_model":"vicuna:33b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q5_K_S":{"mode":"chat","base_model":"vicuna:33b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q6_K":{"mode":"chat","base_model":"vicuna:33b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:33b-q8_0":{"mode":"chat","base_model":"vicuna:33b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-16k":{"mode":"chat","base_model":"vicuna:7b-16k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-fp16":{"mode":"chat","base_model":"vicuna:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q2_K":{"mode":"chat","base_model":"vicuna:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q3_K_L":{"mode":"chat","base_model":"vicuna:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q3_K_M":{"mode":"chat","base_model":"vicuna:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q3_K_S":{"mode":"chat","base_model":"vicuna:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q4_0":{"mode":"chat","base_model":"vicuna:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q4_1":{"mode":"chat","base_model":"vicuna:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q4_K_M":{"mode":"chat","base_model":"vicuna:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q4_K_S":{"mode":"chat","base_model":"vicuna:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q5_0":{"mode":"chat","base_model":"vicuna:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q5_1":{"mode":"chat","base_model":"vicuna:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q5_K_M":{"mode":"chat","base_model":"vicuna:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q5_K_S":{"mode":"chat","base_model":"vicuna:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q6_K":{"mode":"chat","base_model":"vicuna:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-q8_0":{"mode":"chat","base_model":"vicuna:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-fp16":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q2_K":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q3_K_L":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q3_K_M":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q3_K_S":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q4_0":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q4_1":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q4_K_M":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q4_K_S":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q5_0":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q5_1":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q5_K_M":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q5_K_S":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q6_K":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-16k-q8_0":{"mode":"chat","base_model":"vicuna:7b-v1.5-16k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-fp16":{"mode":"chat","base_model":"vicuna:7b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q2_K":{"mode":"chat","base_model":"vicuna:7b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q3_K_L":{"mode":"chat","base_model":"vicuna:7b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q3_K_M":{"mode":"chat","base_model":"vicuna:7b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q3_K_S":{"mode":"chat","base_model":"vicuna:7b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q4_0":{"mode":"chat","base_model":"vicuna:7b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q4_1":{"mode":"chat","base_model":"vicuna:7b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q4_K_M":{"mode":"chat","base_model":"vicuna:7b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q4_K_S":{"mode":"chat","base_model":"vicuna:7b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q5_0":{"mode":"chat","base_model":"vicuna:7b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q5_1":{"mode":"chat","base_model":"vicuna:7b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q5_K_M":{"mode":"chat","base_model":"vicuna:7b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q5_K_S":{"mode":"chat","base_model":"vicuna:7b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q6_K":{"mode":"chat","base_model":"vicuna:7b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"vicuna:7b-v1.5-q8_0":{"mode":"chat","base_model":"vicuna:7b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b":{"mode":"chat","base_model":"mistral-openorca:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-fp16":{"mode":"chat","base_model":"mistral-openorca:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q2_K":{"mode":"chat","base_model":"mistral-openorca:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q3_K_L":{"mode":"chat","base_model":"mistral-openorca:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q3_K_M":{"mode":"chat","base_model":"mistral-openorca:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q3_K_S":{"mode":"chat","base_model":"mistral-openorca:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q4_0":{"mode":"chat","base_model":"mistral-openorca:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q4_1":{"mode":"chat","base_model":"mistral-openorca:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q4_K_M":{"mode":"chat","base_model":"mistral-openorca:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q4_K_S":{"mode":"chat","base_model":"mistral-openorca:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q5_0":{"mode":"chat","base_model":"mistral-openorca:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q5_1":{"mode":"chat","base_model":"mistral-openorca:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q5_K_M":{"mode":"chat","base_model":"mistral-openorca:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q5_K_S":{"mode":"chat","base_model":"mistral-openorca:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q6_K":{"mode":"chat","base_model":"mistral-openorca:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-openorca:7b-q8_0":{"mode":"chat","base_model":"mistral-openorca:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwq:32b-preview-fp16":{"mode":"chat","base_model":"qwq:32b-preview-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwq:32b-preview-q4_K_M":{"mode":"chat","base_model":"qwq:32b-preview-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwq:32b-preview-q8_0":{"mode":"chat","base_model":"qwq:32b-preview-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b":{"mode":"chat","base_model":"llama2-chinese:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b":{"mode":"chat","base_model":"llama2-chinese:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat":{"mode":"chat","base_model":"llama2-chinese:13b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-fp16":{"mode":"chat","base_model":"llama2-chinese:13b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q2_K":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q3_K_L":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q3_K_M":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q3_K_S":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q4_0":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q4_1":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q4_K_M":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q4_K_S":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q5_0":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q5_1":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q5_K_M":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q5_K_S":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q6_K":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:13b-chat-q8_0":{"mode":"chat","base_model":"llama2-chinese:13b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat":{"mode":"chat","base_model":"llama2-chinese:7b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-fp16":{"mode":"chat","base_model":"llama2-chinese:7b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q2_K":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q3_K_L":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q3_K_M":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q3_K_S":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q4_0":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q4_1":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q4_K_M":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q4_K_S":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q5_0":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q5_1":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q5_K_M":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q5_K_S":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q6_K":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama2-chinese:7b-chat-q8_0":{"mode":"chat","base_model":"llama2-chinese:7b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b":{"mode":"chat","base_model":"openchat:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5":{"mode":"chat","base_model":"openchat:7b-v3.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106":{"mode":"chat","base_model":"openchat:7b-v3.5-0106","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-fp16":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q2_K":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q3_K_L":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q3_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q3_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q4_0":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q4_1":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q4_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q4_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q5_0":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q5_1":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q5_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q5_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q6_K":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-0106-q8_0":{"mode":"chat","base_model":"openchat:7b-v3.5-0106-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210":{"mode":"chat","base_model":"openchat:7b-v3.5-1210","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-fp16":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q2_K":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q3_K_L":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q3_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q3_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q4_0":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q4_1":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q4_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q4_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q5_0":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q5_1":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q5_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q5_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q6_K":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-1210-q8_0":{"mode":"chat","base_model":"openchat:7b-v3.5-1210-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-fp16":{"mode":"chat","base_model":"openchat:7b-v3.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q2_K":{"mode":"chat","base_model":"openchat:7b-v3.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q3_K_L":{"mode":"chat","base_model":"openchat:7b-v3.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q3_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q3_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q4_0":{"mode":"chat","base_model":"openchat:7b-v3.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q4_1":{"mode":"chat","base_model":"openchat:7b-v3.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q4_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q4_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q5_0":{"mode":"chat","base_model":"openchat:7b-v3.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q5_1":{"mode":"chat","base_model":"openchat:7b-v3.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q5_K_M":{"mode":"chat","base_model":"openchat:7b-v3.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q5_K_S":{"mode":"chat","base_model":"openchat:7b-v3.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q6_K":{"mode":"chat","base_model":"openchat:7b-v3.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openchat:7b-v3.5-q8_0":{"mode":"chat","base_model":"openchat:7b-v3.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b":{"mode":"chat","base_model":"codegeex4:9b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-fp16":{"mode":"chat","base_model":"codegeex4:9b-all-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q2_K":{"mode":"chat","base_model":"codegeex4:9b-all-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q3_K_L":{"mode":"chat","base_model":"codegeex4:9b-all-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q3_K_M":{"mode":"chat","base_model":"codegeex4:9b-all-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q3_K_S":{"mode":"chat","base_model":"codegeex4:9b-all-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q4_0":{"mode":"chat","base_model":"codegeex4:9b-all-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q4_1":{"mode":"chat","base_model":"codegeex4:9b-all-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q4_K_M":{"mode":"chat","base_model":"codegeex4:9b-all-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q4_K_S":{"mode":"chat","base_model":"codegeex4:9b-all-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q5_0":{"mode":"chat","base_model":"codegeex4:9b-all-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q5_1":{"mode":"chat","base_model":"codegeex4:9b-all-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q5_K_M":{"mode":"chat","base_model":"codegeex4:9b-all-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q5_K_S":{"mode":"chat","base_model":"codegeex4:9b-all-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q6_K":{"mode":"chat","base_model":"codegeex4:9b-all-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codegeex4:9b-all-q8_0":{"mode":"chat","base_model":"codegeex4:9b-all-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b":{"mode":"chat","base_model":"aya:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b":{"mode":"chat","base_model":"aya:35b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23":{"mode":"chat","base_model":"aya:35b-23","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q2_K":{"mode":"chat","base_model":"aya:35b-23-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q3_K_L":{"mode":"chat","base_model":"aya:35b-23-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q3_K_M":{"mode":"chat","base_model":"aya:35b-23-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q3_K_S":{"mode":"chat","base_model":"aya:35b-23-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q4_0":{"mode":"chat","base_model":"aya:35b-23-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q4_1":{"mode":"chat","base_model":"aya:35b-23-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q4_K_M":{"mode":"chat","base_model":"aya:35b-23-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q4_K_S":{"mode":"chat","base_model":"aya:35b-23-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q5_0":{"mode":"chat","base_model":"aya:35b-23-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q5_1":{"mode":"chat","base_model":"aya:35b-23-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q5_K_M":{"mode":"chat","base_model":"aya:35b-23-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q5_K_S":{"mode":"chat","base_model":"aya:35b-23-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q6_K":{"mode":"chat","base_model":"aya:35b-23-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:35b-23-q8_0":{"mode":"chat","base_model":"aya:35b-23-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23":{"mode":"chat","base_model":"aya:8b-23","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q2_K":{"mode":"chat","base_model":"aya:8b-23-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q3_K_L":{"mode":"chat","base_model":"aya:8b-23-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q3_K_M":{"mode":"chat","base_model":"aya:8b-23-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q3_K_S":{"mode":"chat","base_model":"aya:8b-23-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q4_0":{"mode":"chat","base_model":"aya:8b-23-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q4_1":{"mode":"chat","base_model":"aya:8b-23-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q4_K_M":{"mode":"chat","base_model":"aya:8b-23-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q4_K_S":{"mode":"chat","base_model":"aya:8b-23-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q5_0":{"mode":"chat","base_model":"aya:8b-23-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q5_1":{"mode":"chat","base_model":"aya:8b-23-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q5_K_M":{"mode":"chat","base_model":"aya:8b-23-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q5_K_S":{"mode":"chat","base_model":"aya:8b-23-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q6_K":{"mode":"chat","base_model":"aya:8b-23-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya:8b-23-q8_0":{"mode":"chat","base_model":"aya:8b-23-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b":{"mode":"chat","base_model":"codeqwen:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat":{"mode":"chat","base_model":"codeqwen:7b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-fp16":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q2_K":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q3_K_L":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q3_K_M":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q3_K_S":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q4_0":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q4_1":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q4_K_M":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q4_K_S":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q5_0":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q5_1":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q5_K_M":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q5_K_S":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q6_K":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-chat-v1.5-q8_0":{"mode":"chat","base_model":"codeqwen:7b-chat-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-code":{"mode":"chat","base_model":"codeqwen:7b-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-code-v1.5-fp16":{"mode":"chat","base_model":"codeqwen:7b-code-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-code-v1.5-q4_0":{"mode":"chat","base_model":"codeqwen:7b-code-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-code-v1.5-q4_1":{"mode":"chat","base_model":"codeqwen:7b-code-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-code-v1.5-q5_0":{"mode":"chat","base_model":"codeqwen:7b-code-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-code-v1.5-q5_1":{"mode":"chat","base_model":"codeqwen:7b-code-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:7b-code-v1.5-q8_0":{"mode":"chat","base_model":"codeqwen:7b-code-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:chat":{"mode":"chat","base_model":"codeqwen:chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:code":{"mode":"chat","base_model":"codeqwen:code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"codeqwen:v1.5":{"mode":"chat","base_model":"codeqwen:v1.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:v1.5-chat":{"mode":"chat","base_model":"codeqwen:v1.5-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeqwen:v1.5-code":{"mode":"chat","base_model":"codeqwen:v1.5-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b":{"mode":"chat","base_model":"deepseek-llm:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b":{"mode":"chat","base_model":"deepseek-llm:67b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base":{"mode":"chat","base_model":"deepseek-llm:67b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-fp16":{"mode":"chat","base_model":"deepseek-llm:67b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q2_K":{"mode":"chat","base_model":"deepseek-llm:67b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q3_K_L":{"mode":"chat","base_model":"deepseek-llm:67b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q3_K_M":{"mode":"chat","base_model":"deepseek-llm:67b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q3_K_S":{"mode":"chat","base_model":"deepseek-llm:67b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q4_0":{"mode":"chat","base_model":"deepseek-llm:67b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q4_1":{"mode":"chat","base_model":"deepseek-llm:67b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q4_K_M":{"mode":"chat","base_model":"deepseek-llm:67b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q4_K_S":{"mode":"chat","base_model":"deepseek-llm:67b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q5_0":{"mode":"chat","base_model":"deepseek-llm:67b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q5_1":{"mode":"chat","base_model":"deepseek-llm:67b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q5_K_M":{"mode":"chat","base_model":"deepseek-llm:67b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q5_K_S":{"mode":"chat","base_model":"deepseek-llm:67b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q6_K":{"mode":"chat","base_model":"deepseek-llm:67b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-base-q8_0":{"mode":"chat","base_model":"deepseek-llm:67b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat":{"mode":"chat","base_model":"deepseek-llm:67b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-fp16":{"mode":"chat","base_model":"deepseek-llm:67b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q2_K":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q3_K_L":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q3_K_M":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q3_K_S":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q4_0":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q4_1":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q4_K_M":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q4_K_S":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q5_0":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q5_1":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:67b-chat-q5_K_S":{"mode":"chat","base_model":"deepseek-llm:67b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base":{"mode":"chat","base_model":"deepseek-llm:7b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-fp16":{"mode":"chat","base_model":"deepseek-llm:7b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q2_K":{"mode":"chat","base_model":"deepseek-llm:7b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q3_K_L":{"mode":"chat","base_model":"deepseek-llm:7b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q3_K_M":{"mode":"chat","base_model":"deepseek-llm:7b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q3_K_S":{"mode":"chat","base_model":"deepseek-llm:7b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q4_0":{"mode":"chat","base_model":"deepseek-llm:7b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q4_1":{"mode":"chat","base_model":"deepseek-llm:7b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q4_K_M":{"mode":"chat","base_model":"deepseek-llm:7b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q4_K_S":{"mode":"chat","base_model":"deepseek-llm:7b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q5_0":{"mode":"chat","base_model":"deepseek-llm:7b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q5_1":{"mode":"chat","base_model":"deepseek-llm:7b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q5_K_M":{"mode":"chat","base_model":"deepseek-llm:7b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q5_K_S":{"mode":"chat","base_model":"deepseek-llm:7b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q6_K":{"mode":"chat","base_model":"deepseek-llm:7b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-base-q8_0":{"mode":"chat","base_model":"deepseek-llm:7b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat":{"mode":"chat","base_model":"deepseek-llm:7b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-fp16":{"mode":"chat","base_model":"deepseek-llm:7b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q2_K":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q3_K_L":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q3_K_M":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q3_K_S":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q4_0":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q4_1":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q4_K_M":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q4_K_S":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q5_0":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q5_1":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q5_K_M":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q5_K_S":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q6_K":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-llm:7b-chat-q8_0":{"mode":"chat","base_model":"deepseek-llm:7b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b":{"mode":"chat","base_model":"deepseek-v2:16b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b":{"mode":"chat","base_model":"deepseek-v2:236b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-fp16":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q2_K":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q3_K_L":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q3_K_M":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q3_K_S":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q4_0":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q4_1":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q4_K_M":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q4_K_S":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q5_0":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q5_1":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q5_K_M":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q5_K_S":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q6_K":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:16b-lite-chat-q8_0":{"mode":"chat","base_model":"deepseek-v2:16b-lite-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-fp16":{"mode":"chat","base_model":"deepseek-v2:236b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q2_K":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q3_K_L":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q3_K_M":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q3_K_S":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q4_0":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q4_1":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q4_K_M":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q4_K_S":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q5_0":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q5_1":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q5_K_M":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q5_K_S":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q6_K":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:236b-chat-q8_0":{"mode":"chat","base_model":"deepseek-v2:236b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2:lite":{"mode":"chat","base_model":"deepseek-v2:lite","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b":{"mode":"chat","base_model":"mistral-large:123b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-fp16":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q2_K":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q3_K_L":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q3_K_M":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q3_K_S":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q4_0":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q4_1":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q4_K_M":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q4_K_S":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q5_0":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q5_1":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q5_K_M":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q5_K_S":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q6_K":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2407-q8_0":{"mode":"chat","base_model":"mistral-large:123b-instruct-2407-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-fp16":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q2_K":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q3_K_L":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q3_K_M":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q3_K_S":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q4_0":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q4_1":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q4_K_M":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q4_K_S":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q5_0":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q5_1":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q5_K_M":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q5_K_S":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q6_K":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral-large:123b-instruct-2411-q8_0":{"mode":"chat","base_model":"mistral-large:123b-instruct-2411-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b":{"mode":"chat","base_model":"glm4:9b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-fp16":{"mode":"chat","base_model":"glm4:9b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q2_K":{"mode":"chat","base_model":"glm4:9b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q3_K_L":{"mode":"chat","base_model":"glm4:9b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q3_K_M":{"mode":"chat","base_model":"glm4:9b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q3_K_S":{"mode":"chat","base_model":"glm4:9b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q4_0":{"mode":"chat","base_model":"glm4:9b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q4_1":{"mode":"chat","base_model":"glm4:9b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q4_K_M":{"mode":"chat","base_model":"glm4:9b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q4_K_S":{"mode":"chat","base_model":"glm4:9b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q5_0":{"mode":"chat","base_model":"glm4:9b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q5_1":{"mode":"chat","base_model":"glm4:9b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q5_K_M":{"mode":"chat","base_model":"glm4:9b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q5_K_S":{"mode":"chat","base_model":"glm4:9b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q6_K":{"mode":"chat","base_model":"glm4:9b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-chat-q8_0":{"mode":"chat","base_model":"glm4:9b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-fp16":{"mode":"chat","base_model":"glm4:9b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q2_K":{"mode":"chat","base_model":"glm4:9b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q3_K_L":{"mode":"chat","base_model":"glm4:9b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q3_K_M":{"mode":"chat","base_model":"glm4:9b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q3_K_S":{"mode":"chat","base_model":"glm4:9b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q4_0":{"mode":"chat","base_model":"glm4:9b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q4_1":{"mode":"chat","base_model":"glm4:9b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q4_K_M":{"mode":"chat","base_model":"glm4:9b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q4_K_S":{"mode":"chat","base_model":"glm4:9b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q5_0":{"mode":"chat","base_model":"glm4:9b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q5_1":{"mode":"chat","base_model":"glm4:9b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q5_K_M":{"mode":"chat","base_model":"glm4:9b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q5_K_S":{"mode":"chat","base_model":"glm4:9b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q6_K":{"mode":"chat","base_model":"glm4:9b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"glm4:9b-text-q8_0":{"mode":"chat","base_model":"glm4:9b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b":{"mode":"chat","base_model":"nous-hermes2:10.7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b":{"mode":"chat","base_model":"nous-hermes2:34b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-fp16":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q2_K":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q3_K_L":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q3_K_M":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q3_K_S":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q4_0":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q4_1":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q4_K_M":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q4_K_S":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q5_0":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q5_1":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q5_K_M":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q5_K_S":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q6_K":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:10.7b-solar-q8_0":{"mode":"chat","base_model":"nous-hermes2:10.7b-solar-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-fp16":{"mode":"chat","base_model":"nous-hermes2:34b-yi-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q2_K":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q3_K_L":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q3_K_M":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q3_K_S":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q4_0":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q4_1":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q4_K_M":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q4_K_S":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q5_0":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q5_1":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q5_K_M":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q5_K_S":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q6_K":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2:34b-yi-q8_0":{"mode":"chat","base_model":"nous-hermes2:34b-yi-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code":{"mode":"chat","base_model":"stable-code:3b-code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-fp16":{"mode":"chat","base_model":"stable-code:3b-code-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q2_K":{"mode":"chat","base_model":"stable-code:3b-code-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q3_K_L":{"mode":"chat","base_model":"stable-code:3b-code-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q3_K_M":{"mode":"chat","base_model":"stable-code:3b-code-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q3_K_S":{"mode":"chat","base_model":"stable-code:3b-code-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q4_0":{"mode":"chat","base_model":"stable-code:3b-code-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q4_1":{"mode":"chat","base_model":"stable-code:3b-code-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q4_K_M":{"mode":"chat","base_model":"stable-code:3b-code-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q4_K_S":{"mode":"chat","base_model":"stable-code:3b-code-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q5_0":{"mode":"chat","base_model":"stable-code:3b-code-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q5_1":{"mode":"chat","base_model":"stable-code:3b-code-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q5_K_M":{"mode":"chat","base_model":"stable-code:3b-code-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q5_K_S":{"mode":"chat","base_model":"stable-code:3b-code-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q6_K":{"mode":"chat","base_model":"stable-code:3b-code-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-code-q8_0":{"mode":"chat","base_model":"stable-code:3b-code-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct":{"mode":"chat","base_model":"stable-code:3b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-fp16":{"mode":"chat","base_model":"stable-code:3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q2_K":{"mode":"chat","base_model":"stable-code:3b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q3_K_L":{"mode":"chat","base_model":"stable-code:3b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q3_K_M":{"mode":"chat","base_model":"stable-code:3b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q3_K_S":{"mode":"chat","base_model":"stable-code:3b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q4_0":{"mode":"chat","base_model":"stable-code:3b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q4_1":{"mode":"chat","base_model":"stable-code:3b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q4_K_M":{"mode":"chat","base_model":"stable-code:3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q4_K_S":{"mode":"chat","base_model":"stable-code:3b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q5_0":{"mode":"chat","base_model":"stable-code:3b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q5_1":{"mode":"chat","base_model":"stable-code:3b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q5_K_M":{"mode":"chat","base_model":"stable-code:3b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q5_K_S":{"mode":"chat","base_model":"stable-code:3b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q6_K":{"mode":"chat","base_model":"stable-code:3b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:3b-instruct-q8_0":{"mode":"chat","base_model":"stable-code:3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:code":{"mode":"chat","base_model":"stable-code:code","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-code:instruct":{"mode":"chat","base_model":"stable-code:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-fp16":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q2_K":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q3_K_L":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q3_K_M":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q3_K_S":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q4_0":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q4_1":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q4_K_M":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q4_K_S":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q5_0":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q5_1":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q5_K_M":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q5_K_S":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q6_K":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2-q8_0":{"mode":"chat","base_model":"openhermes:7b-mistral-v2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-fp16":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q2_K":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q3_K_L":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q3_K_M":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q3_K_S":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q4_0":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q4_1":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q4_K_M":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q4_K_S":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q5_0":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q5_1":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q5_K_M":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q5_K_S":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q6_K":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-mistral-v2.5-q8_0":{"mode":"chat","base_model":"openhermes:7b-mistral-v2.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-v2":{"mode":"chat","base_model":"openhermes:7b-v2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:7b-v2.5":{"mode":"chat","base_model":"openhermes:7b-v2.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:v2":{"mode":"chat","base_model":"openhermes:v2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openhermes:v2.5":{"mode":"chat","base_model":"openhermes:v2.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b":{"mode":"chat","base_model":"qwen2-math:1.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b":{"mode":"chat","base_model":"qwen2-math:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b":{"mode":"chat","base_model":"qwen2-math:72b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-fp16":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q2_K":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q4_0":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q4_1":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q5_0":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q5_1":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q6_K":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:1.5b-instruct-q8_0":{"mode":"chat","base_model":"qwen2-math:1.5b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct":{"mode":"chat","base_model":"qwen2-math:72b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-fp16":{"mode":"chat","base_model":"qwen2-math:72b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q2_K":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q4_0":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q4_1":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q5_0":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q5_1":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q6_K":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:72b-instruct-q8_0":{"mode":"chat","base_model":"qwen2-math:72b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct":{"mode":"chat","base_model":"qwen2-math:7b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-fp16":{"mode":"chat","base_model":"qwen2-math:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q2_K":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q3_K_L":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q3_K_M":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q3_K_S":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q4_0":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q4_1":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q4_K_M":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q4_K_S":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q5_0":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q5_1":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q5_K_M":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q5_K_S":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q6_K":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"qwen2-math:7b-instruct-q8_0":{"mode":"chat","base_model":"qwen2-math:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b":{"mode":"chat","base_model":"tinydolphin:1.1b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-fp16":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q2_K":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q3_K_L":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q3_K_M":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q3_K_S":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q4_0":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q4_1":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q4_K_M":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q4_K_S":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q5_0":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q5_1":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q5_K_M":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q5_K_S":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q6_K":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:1.1b-v2.8-q8_0":{"mode":"chat","base_model":"tinydolphin:1.1b-v2.8-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tinydolphin:v2.8":{"mode":"chat","base_model":"tinydolphin:v2.8","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b":{"mode":"chat","base_model":"command-r-plus:104b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-fp16":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q2_K":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q3_K_L":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q3_K_M":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q3_K_S":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q4_0":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q4_1":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q4_K_M":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q4_K_S":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q5_0":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q5_1":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q5_K_M":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q5_K_S":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q6_K":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-08-2024-q8_0":{"mode":"chat","base_model":"command-r-plus:104b-08-2024-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-fp16":{"mode":"chat","base_model":"command-r-plus:104b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-q2_K":{"mode":"chat","base_model":"command-r-plus:104b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-q4_0":{"mode":"chat","base_model":"command-r-plus:104b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r-plus:104b-q8_0":{"mode":"chat","base_model":"command-r-plus:104b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b":{"mode":"chat","base_model":"wizardcoder:33b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python":{"mode":"chat","base_model":"wizardcoder:13b-python","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-fp16":{"mode":"chat","base_model":"wizardcoder:13b-python-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q2_K":{"mode":"chat","base_model":"wizardcoder:13b-python-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q3_K_L":{"mode":"chat","base_model":"wizardcoder:13b-python-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q3_K_M":{"mode":"chat","base_model":"wizardcoder:13b-python-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q3_K_S":{"mode":"chat","base_model":"wizardcoder:13b-python-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q4_0":{"mode":"chat","base_model":"wizardcoder:13b-python-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q4_1":{"mode":"chat","base_model":"wizardcoder:13b-python-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q4_K_M":{"mode":"chat","base_model":"wizardcoder:13b-python-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q4_K_S":{"mode":"chat","base_model":"wizardcoder:13b-python-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q5_0":{"mode":"chat","base_model":"wizardcoder:13b-python-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q5_1":{"mode":"chat","base_model":"wizardcoder:13b-python-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q5_K_M":{"mode":"chat","base_model":"wizardcoder:13b-python-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q5_K_S":{"mode":"chat","base_model":"wizardcoder:13b-python-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q6_K":{"mode":"chat","base_model":"wizardcoder:13b-python-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:13b-python-q8_0":{"mode":"chat","base_model":"wizardcoder:13b-python-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1":{"mode":"chat","base_model":"wizardcoder:33b-v1.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-fp16":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q2_K":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q3_K_L":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q3_K_M":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q3_K_S":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q4_0":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q4_1":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q4_K_M":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q4_K_S":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q5_0":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q5_1":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q5_K_M":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q5_K_S":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q6_K":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:33b-v1.1-q8_0":{"mode":"chat","base_model":"wizardcoder:33b-v1.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python":{"mode":"chat","base_model":"wizardcoder:34b-python","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-fp16":{"mode":"chat","base_model":"wizardcoder:34b-python-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q2_K":{"mode":"chat","base_model":"wizardcoder:34b-python-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q3_K_L":{"mode":"chat","base_model":"wizardcoder:34b-python-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q3_K_M":{"mode":"chat","base_model":"wizardcoder:34b-python-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q3_K_S":{"mode":"chat","base_model":"wizardcoder:34b-python-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q4_0":{"mode":"chat","base_model":"wizardcoder:34b-python-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q4_1":{"mode":"chat","base_model":"wizardcoder:34b-python-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q4_K_M":{"mode":"chat","base_model":"wizardcoder:34b-python-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q4_K_S":{"mode":"chat","base_model":"wizardcoder:34b-python-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q5_0":{"mode":"chat","base_model":"wizardcoder:34b-python-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q5_1":{"mode":"chat","base_model":"wizardcoder:34b-python-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q5_K_M":{"mode":"chat","base_model":"wizardcoder:34b-python-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q5_K_S":{"mode":"chat","base_model":"wizardcoder:34b-python-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q6_K":{"mode":"chat","base_model":"wizardcoder:34b-python-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:34b-python-q8_0":{"mode":"chat","base_model":"wizardcoder:34b-python-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python":{"mode":"chat","base_model":"wizardcoder:7b-python","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-fp16":{"mode":"chat","base_model":"wizardcoder:7b-python-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q2_K":{"mode":"chat","base_model":"wizardcoder:7b-python-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q3_K_L":{"mode":"chat","base_model":"wizardcoder:7b-python-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q3_K_M":{"mode":"chat","base_model":"wizardcoder:7b-python-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q3_K_S":{"mode":"chat","base_model":"wizardcoder:7b-python-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q4_0":{"mode":"chat","base_model":"wizardcoder:7b-python-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q4_1":{"mode":"chat","base_model":"wizardcoder:7b-python-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q4_K_M":{"mode":"chat","base_model":"wizardcoder:7b-python-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q4_K_S":{"mode":"chat","base_model":"wizardcoder:7b-python-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q5_0":{"mode":"chat","base_model":"wizardcoder:7b-python-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q5_1":{"mode":"chat","base_model":"wizardcoder:7b-python-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q5_K_M":{"mode":"chat","base_model":"wizardcoder:7b-python-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q5_K_S":{"mode":"chat","base_model":"wizardcoder:7b-python-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q6_K":{"mode":"chat","base_model":"wizardcoder:7b-python-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:7b-python-q8_0":{"mode":"chat","base_model":"wizardcoder:7b-python-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardcoder:python":{"mode":"chat","base_model":"wizardcoder:python","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b":{"mode":"chat","base_model":"moondream:1.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-fp16":{"mode":"chat","base_model":"moondream:1.8b-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q2_K":{"mode":"chat","base_model":"moondream:1.8b-v2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q3_K_L":{"mode":"chat","base_model":"moondream:1.8b-v2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q3_K_M":{"mode":"chat","base_model":"moondream:1.8b-v2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q3_K_S":{"mode":"chat","base_model":"moondream:1.8b-v2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q4_0":{"mode":"chat","base_model":"moondream:1.8b-v2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q4_1":{"mode":"chat","base_model":"moondream:1.8b-v2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q4_K_M":{"mode":"chat","base_model":"moondream:1.8b-v2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q4_K_S":{"mode":"chat","base_model":"moondream:1.8b-v2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q5_0":{"mode":"chat","base_model":"moondream:1.8b-v2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q5_1":{"mode":"chat","base_model":"moondream:1.8b-v2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q5_K_M":{"mode":"chat","base_model":"moondream:1.8b-v2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q5_K_S":{"mode":"chat","base_model":"moondream:1.8b-v2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q6_K":{"mode":"chat","base_model":"moondream:1.8b-v2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:1.8b-v2-q8_0":{"mode":"chat","base_model":"moondream:1.8b-v2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"moondream:v2":{"mode":"chat","base_model":"moondream:v2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b":{"mode":"chat","base_model":"bakllava:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-fp16":{"mode":"chat","base_model":"bakllava:7b-v1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q2_K":{"mode":"chat","base_model":"bakllava:7b-v1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q3_K_L":{"mode":"chat","base_model":"bakllava:7b-v1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q3_K_M":{"mode":"chat","base_model":"bakllava:7b-v1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q3_K_S":{"mode":"chat","base_model":"bakllava:7b-v1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q4_0":{"mode":"chat","base_model":"bakllava:7b-v1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q4_1":{"mode":"chat","base_model":"bakllava:7b-v1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q4_K_M":{"mode":"chat","base_model":"bakllava:7b-v1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q4_K_S":{"mode":"chat","base_model":"bakllava:7b-v1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q5_0":{"mode":"chat","base_model":"bakllava:7b-v1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q5_1":{"mode":"chat","base_model":"bakllava:7b-v1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q5_K_M":{"mode":"chat","base_model":"bakllava:7b-v1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q5_K_S":{"mode":"chat","base_model":"bakllava:7b-v1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q6_K":{"mode":"chat","base_model":"bakllava:7b-v1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bakllava:7b-v1-q8_0":{"mode":"chat","base_model":"bakllava:7b-v1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b":{"mode":"chat","base_model":"stablelm2:1.6b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b":{"mode":"chat","base_model":"stablelm2:12b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat":{"mode":"chat","base_model":"stablelm2:1.6b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-fp16":{"mode":"chat","base_model":"stablelm2:1.6b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q2_K":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q3_K_L":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q3_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q3_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q4_0":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q4_1":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q4_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q4_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q5_0":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q5_1":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q5_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q5_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q6_K":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-chat-q8_0":{"mode":"chat","base_model":"stablelm2:1.6b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-fp16":{"mode":"chat","base_model":"stablelm2:1.6b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q2_K":{"mode":"chat","base_model":"stablelm2:1.6b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q3_K_L":{"mode":"chat","base_model":"stablelm2:1.6b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q3_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q3_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q4_0":{"mode":"chat","base_model":"stablelm2:1.6b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q4_1":{"mode":"chat","base_model":"stablelm2:1.6b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q4_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q4_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q5_0":{"mode":"chat","base_model":"stablelm2:1.6b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q5_1":{"mode":"chat","base_model":"stablelm2:1.6b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q5_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q5_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q6_K":{"mode":"chat","base_model":"stablelm2:1.6b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-q8_0":{"mode":"chat","base_model":"stablelm2:1.6b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-fp16":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q2_K":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q3_K_L":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q3_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q3_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q4_0":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q4_1":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q4_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q4_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q5_0":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q5_1":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q5_K_M":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q5_K_S":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q6_K":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:1.6b-zephyr-q8_0":{"mode":"chat","base_model":"stablelm2:1.6b-zephyr-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat":{"mode":"chat","base_model":"stablelm2:12b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-fp16":{"mode":"chat","base_model":"stablelm2:12b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q2_K":{"mode":"chat","base_model":"stablelm2:12b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q3_K_L":{"mode":"chat","base_model":"stablelm2:12b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q3_K_M":{"mode":"chat","base_model":"stablelm2:12b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q3_K_S":{"mode":"chat","base_model":"stablelm2:12b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q4_0":{"mode":"chat","base_model":"stablelm2:12b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q4_1":{"mode":"chat","base_model":"stablelm2:12b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q4_K_M":{"mode":"chat","base_model":"stablelm2:12b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q4_K_S":{"mode":"chat","base_model":"stablelm2:12b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q5_0":{"mode":"chat","base_model":"stablelm2:12b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q5_1":{"mode":"chat","base_model":"stablelm2:12b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q5_K_M":{"mode":"chat","base_model":"stablelm2:12b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q5_K_S":{"mode":"chat","base_model":"stablelm2:12b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q6_K":{"mode":"chat","base_model":"stablelm2:12b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-chat-q8_0":{"mode":"chat","base_model":"stablelm2:12b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-fp16":{"mode":"chat","base_model":"stablelm2:12b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q2_K":{"mode":"chat","base_model":"stablelm2:12b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q3_K_L":{"mode":"chat","base_model":"stablelm2:12b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q3_K_M":{"mode":"chat","base_model":"stablelm2:12b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q3_K_S":{"mode":"chat","base_model":"stablelm2:12b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q4_0":{"mode":"chat","base_model":"stablelm2:12b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q4_1":{"mode":"chat","base_model":"stablelm2:12b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q4_K_M":{"mode":"chat","base_model":"stablelm2:12b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q4_K_S":{"mode":"chat","base_model":"stablelm2:12b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q5_0":{"mode":"chat","base_model":"stablelm2:12b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q5_1":{"mode":"chat","base_model":"stablelm2:12b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q5_K_M":{"mode":"chat","base_model":"stablelm2:12b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q5_K_S":{"mode":"chat","base_model":"stablelm2:12b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q6_K":{"mode":"chat","base_model":"stablelm2:12b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-q8_0":{"mode":"chat","base_model":"stablelm2:12b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:12b-text":{"mode":"chat","base_model":"stablelm2:12b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:chat":{"mode":"chat","base_model":"stablelm2:chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm2:zephyr":{"mode":"chat","base_model":"stablelm2:zephyr","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b":{"mode":"chat","base_model":"neural-chat:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1":{"mode":"chat","base_model":"neural-chat:7b-v3.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-fp16":{"mode":"chat","base_model":"neural-chat:7b-v3.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q2_K":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q3_K_L":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q3_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q3_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q4_0":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q4_1":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q4_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q4_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q5_0":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q5_1":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q5_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q5_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q6_K":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.1-q8_0":{"mode":"chat","base_model":"neural-chat:7b-v3.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2":{"mode":"chat","base_model":"neural-chat:7b-v3.2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-fp16":{"mode":"chat","base_model":"neural-chat:7b-v3.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q2_K":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q3_K_L":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q3_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q3_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q4_0":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q4_1":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q4_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q4_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q5_0":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q5_1":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q5_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q5_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q6_K":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.2-q8_0":{"mode":"chat","base_model":"neural-chat:7b-v3.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3":{"mode":"chat","base_model":"neural-chat:7b-v3.3","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-fp16":{"mode":"chat","base_model":"neural-chat:7b-v3.3-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q2_K":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q3_K_L":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q3_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q3_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q4_0":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q4_1":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q4_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q4_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q5_0":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q5_1":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q5_K_M":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q5_K_S":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q6_K":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"neural-chat:7b-v3.3-q8_0":{"mode":"chat","base_model":"neural-chat:7b-v3.3-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b":{"mode":"chat","base_model":"reflection:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-fp16":{"mode":"chat","base_model":"reflection:70b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q2_K":{"mode":"chat","base_model":"reflection:70b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q3_K_L":{"mode":"chat","base_model":"reflection:70b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q3_K_M":{"mode":"chat","base_model":"reflection:70b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q3_K_S":{"mode":"chat","base_model":"reflection:70b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q4_0":{"mode":"chat","base_model":"reflection:70b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q4_1":{"mode":"chat","base_model":"reflection:70b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q4_K_M":{"mode":"chat","base_model":"reflection:70b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q4_K_S":{"mode":"chat","base_model":"reflection:70b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q5_0":{"mode":"chat","base_model":"reflection:70b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q5_1":{"mode":"chat","base_model":"reflection:70b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q5_K_M":{"mode":"chat","base_model":"reflection:70b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q5_K_S":{"mode":"chat","base_model":"reflection:70b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q6_K":{"mode":"chat","base_model":"reflection:70b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reflection:70b-q8_0":{"mode":"chat","base_model":"reflection:70b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b":{"mode":"chat","base_model":"wizard-math:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b":{"mode":"chat","base_model":"wizard-math:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b":{"mode":"chat","base_model":"wizard-math:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-fp16":{"mode":"chat","base_model":"wizard-math:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q2_K":{"mode":"chat","base_model":"wizard-math:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q3_K_L":{"mode":"chat","base_model":"wizard-math:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q3_K_M":{"mode":"chat","base_model":"wizard-math:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q3_K_S":{"mode":"chat","base_model":"wizard-math:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q4_0":{"mode":"chat","base_model":"wizard-math:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q4_1":{"mode":"chat","base_model":"wizard-math:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q4_K_M":{"mode":"chat","base_model":"wizard-math:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q4_K_S":{"mode":"chat","base_model":"wizard-math:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q5_0":{"mode":"chat","base_model":"wizard-math:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q5_1":{"mode":"chat","base_model":"wizard-math:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q5_K_M":{"mode":"chat","base_model":"wizard-math:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q5_K_S":{"mode":"chat","base_model":"wizard-math:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q6_K":{"mode":"chat","base_model":"wizard-math:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:13b-q8_0":{"mode":"chat","base_model":"wizard-math:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-fp16":{"mode":"chat","base_model":"wizard-math:70b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q2_K":{"mode":"chat","base_model":"wizard-math:70b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q3_K_L":{"mode":"chat","base_model":"wizard-math:70b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q3_K_M":{"mode":"chat","base_model":"wizard-math:70b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q3_K_S":{"mode":"chat","base_model":"wizard-math:70b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q4_0":{"mode":"chat","base_model":"wizard-math:70b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q4_1":{"mode":"chat","base_model":"wizard-math:70b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q4_K_M":{"mode":"chat","base_model":"wizard-math:70b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q4_K_S":{"mode":"chat","base_model":"wizard-math:70b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q5_0":{"mode":"chat","base_model":"wizard-math:70b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q5_1":{"mode":"chat","base_model":"wizard-math:70b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q5_K_M":{"mode":"chat","base_model":"wizard-math:70b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q5_K_S":{"mode":"chat","base_model":"wizard-math:70b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q6_K":{"mode":"chat","base_model":"wizard-math:70b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:70b-q8_0":{"mode":"chat","base_model":"wizard-math:70b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-fp16":{"mode":"chat","base_model":"wizard-math:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q2_K":{"mode":"chat","base_model":"wizard-math:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q3_K_L":{"mode":"chat","base_model":"wizard-math:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q3_K_M":{"mode":"chat","base_model":"wizard-math:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q3_K_S":{"mode":"chat","base_model":"wizard-math:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q4_0":{"mode":"chat","base_model":"wizard-math:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q4_1":{"mode":"chat","base_model":"wizard-math:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q4_K_M":{"mode":"chat","base_model":"wizard-math:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q4_K_S":{"mode":"chat","base_model":"wizard-math:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q5_0":{"mode":"chat","base_model":"wizard-math:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q5_1":{"mode":"chat","base_model":"wizard-math:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q5_K_M":{"mode":"chat","base_model":"wizard-math:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q5_K_S":{"mode":"chat","base_model":"wizard-math:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q6_K":{"mode":"chat","base_model":"wizard-math:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-q8_0":{"mode":"chat","base_model":"wizard-math:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-fp16":{"mode":"chat","base_model":"wizard-math:7b-v1.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q2_K":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q3_K_L":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q3_K_M":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q3_K_S":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q4_0":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q4_1":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q4_K_M":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q4_K_S":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q5_0":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q5_1":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q5_K_M":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q5_K_S":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q6_K":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-math:7b-v1.1-q8_0":{"mode":"chat","base_model":"wizard-math:7b-v1.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:1048k":{"mode":"chat","base_model":"llama3-gradient:1048k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b":{"mode":"chat","base_model":"llama3-gradient:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b":{"mode":"chat","base_model":"llama3-gradient:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-fp16":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q2_K":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q3_K_L":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q3_K_M":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q3_K_S":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q4_0":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q4_1":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q4_K_M":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q4_K_S":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q5_0":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q5_1":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q5_K_M":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q5_K_S":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q6_K":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:70b-instruct-1048k-q8_0":{"mode":"chat","base_model":"llama3-gradient:70b-instruct-1048k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-fp16":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q2_K":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q3_K_L":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q3_K_M":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q3_K_S":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q4_0":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q4_1":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q4_K_M":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q4_K_S":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q5_0":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q5_1":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q5_K_M":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q5_K_S":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q6_K":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:8b-instruct-1048k-q8_0":{"mode":"chat","base_model":"llama3-gradient:8b-instruct-1048k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-gradient:instruct":{"mode":"chat","base_model":"llama3-gradient:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b":{"mode":"chat","base_model":"llama3-chatqa:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b":{"mode":"chat","base_model":"llama3-chatqa:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-fp16":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q2_K":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q3_K_L":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q3_K_M":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q3_K_S":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q4_0":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q4_1":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q4_K_M":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q4_K_S":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q5_0":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q5_1":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q5_K_M":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q5_K_S":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q6_K":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:70b-v1.5-q8_0":{"mode":"chat","base_model":"llama3-chatqa:70b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-fp16":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q2_K":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q3_K_L":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q3_K_M":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q3_K_S":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q4_0":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q4_1":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q4_K_M":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q4_K_S":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q5_0":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q5_1":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q5_K_M":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q5_K_S":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q6_K":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-chatqa:8b-v1.5-q8_0":{"mode":"chat","base_model":"llama3-chatqa:8b-v1.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b":{"mode":"chat","base_model":"sqlcoder:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b":{"mode":"chat","base_model":"sqlcoder:15b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-fp16":{"mode":"chat","base_model":"sqlcoder:15b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q2_K":{"mode":"chat","base_model":"sqlcoder:15b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q3_K_L":{"mode":"chat","base_model":"sqlcoder:15b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q3_K_M":{"mode":"chat","base_model":"sqlcoder:15b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q3_K_S":{"mode":"chat","base_model":"sqlcoder:15b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q4_0":{"mode":"chat","base_model":"sqlcoder:15b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q4_1":{"mode":"chat","base_model":"sqlcoder:15b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q4_K_M":{"mode":"chat","base_model":"sqlcoder:15b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q4_K_S":{"mode":"chat","base_model":"sqlcoder:15b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q5_0":{"mode":"chat","base_model":"sqlcoder:15b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q5_1":{"mode":"chat","base_model":"sqlcoder:15b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q5_K_M":{"mode":"chat","base_model":"sqlcoder:15b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q5_K_S":{"mode":"chat","base_model":"sqlcoder:15b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q6_K":{"mode":"chat","base_model":"sqlcoder:15b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:15b-q8_0":{"mode":"chat","base_model":"sqlcoder:15b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-fp16":{"mode":"chat","base_model":"sqlcoder:70b-alpha-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q2_K":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q3_K_L":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q3_K_M":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q3_K_S":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q4_0":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q4_1":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q4_K_M":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q4_K_S":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q5_0":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q5_1":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q5_K_M":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q5_K_S":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q6_K":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:70b-alpha-q8_0":{"mode":"chat","base_model":"sqlcoder:70b-alpha-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-fp16":{"mode":"chat","base_model":"sqlcoder:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q2_K":{"mode":"chat","base_model":"sqlcoder:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q3_K_L":{"mode":"chat","base_model":"sqlcoder:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q3_K_M":{"mode":"chat","base_model":"sqlcoder:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q3_K_S":{"mode":"chat","base_model":"sqlcoder:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q4_0":{"mode":"chat","base_model":"sqlcoder:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q4_1":{"mode":"chat","base_model":"sqlcoder:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q4_K_M":{"mode":"chat","base_model":"sqlcoder:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q4_K_S":{"mode":"chat","base_model":"sqlcoder:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q5_0":{"mode":"chat","base_model":"sqlcoder:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q5_1":{"mode":"chat","base_model":"sqlcoder:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q5_K_M":{"mode":"chat","base_model":"sqlcoder:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q5_K_S":{"mode":"chat","base_model":"sqlcoder:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q6_K":{"mode":"chat","base_model":"sqlcoder:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sqlcoder:7b-q8_0":{"mode":"chat","base_model":"sqlcoder:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bge-large:335m":{"mode":"chat","base_model":"bge-large:335m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bge-large:335m-en-v1.5-fp16":{"mode":"chat","base_model":"bge-large:335m-en-v1.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b":{"mode":"chat","base_model":"xwinlm:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b":{"mode":"chat","base_model":"xwinlm:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1":{"mode":"chat","base_model":"xwinlm:13b-v0.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-fp16":{"mode":"chat","base_model":"xwinlm:13b-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q2_K":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q3_K_L":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q3_K_M":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q3_K_S":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q4_0":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q4_1":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q4_K_M":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q4_K_S":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q5_0":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q5_1":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q5_K_M":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q5_K_S":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q6_K":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.1-q8_0":{"mode":"chat","base_model":"xwinlm:13b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2":{"mode":"chat","base_model":"xwinlm:13b-v0.2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-fp16":{"mode":"chat","base_model":"xwinlm:13b-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q2_K":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q3_K_L":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q3_K_M":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q3_K_S":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q4_0":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q4_1":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q4_K_M":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q4_K_S":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q5_0":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q5_1":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q5_K_M":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q5_K_S":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q6_K":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:13b-v0.2-q8_0":{"mode":"chat","base_model":"xwinlm:13b-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1":{"mode":"chat","base_model":"xwinlm:70b-v0.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-fp16":{"mode":"chat","base_model":"xwinlm:70b-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q2_K":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q3_K_L":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q3_K_M":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q3_K_S":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q4_0":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q4_1":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q4_K_M":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q4_K_S":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q5_0":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q5_1":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q5_K_S":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q6_K":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:70b-v0.1-q8_0":{"mode":"chat","base_model":"xwinlm:70b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1":{"mode":"chat","base_model":"xwinlm:7b-v0.1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-fp16":{"mode":"chat","base_model":"xwinlm:7b-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q2_K":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q3_K_L":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q3_K_M":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q3_K_S":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q4_0":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q4_1":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q4_K_M":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q4_K_S":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q5_0":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q5_1":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q5_K_M":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q5_K_S":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q6_K":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.1-q8_0":{"mode":"chat","base_model":"xwinlm:7b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2":{"mode":"chat","base_model":"xwinlm:7b-v0.2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-fp16":{"mode":"chat","base_model":"xwinlm:7b-v0.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q2_K":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q3_K_L":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q3_K_S":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q4_0":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q4_1":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q4_K_M":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q4_K_S":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q5_0":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q5_K_M":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q5_K_S":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q6_K":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"xwinlm:7b-v0.2-q8_0":{"mode":"chat","base_model":"xwinlm:7b-v0.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b":{"mode":"chat","base_model":"dolphincoder:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b":{"mode":"chat","base_model":"dolphincoder:15b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-fp16":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q2_K":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q3_K_L":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q3_K_M":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q3_K_S":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q4_0":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q4_1":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q4_K_M":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q4_K_S":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q5_0":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q5_1":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q5_K_M":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q5_K_S":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q6_K":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:15b-starcoder2-q8_0":{"mode":"chat","base_model":"dolphincoder:15b-starcoder2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-fp16":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q2_K":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q3_K_L":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q3_K_M":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q3_K_S":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q4_0":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q4_1":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q4_K_M":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q4_K_S":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q5_0":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q5_1":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q5_K_M":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q5_K_S":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q6_K":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphincoder:7b-starcoder2-q8_0":{"mode":"chat","base_model":"dolphincoder:7b-starcoder2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b":{"mode":"chat","base_model":"nous-hermes:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b":{"mode":"chat","base_model":"nous-hermes:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-fp16":{"mode":"chat","base_model":"nous-hermes:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2":{"mode":"chat","base_model":"nous-hermes:13b-llama2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-fp16":{"mode":"chat","base_model":"nous-hermes:13b-llama2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q2_K":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q3_K_L":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q3_K_M":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q3_K_S":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q4_0":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q4_1":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q4_K_M":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q4_K_S":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q5_0":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q5_1":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q5_K_M":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q5_K_S":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q6_K":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-llama2-q8_0":{"mode":"chat","base_model":"nous-hermes:13b-llama2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q2_K":{"mode":"chat","base_model":"nous-hermes:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q3_K_L":{"mode":"chat","base_model":"nous-hermes:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q3_K_M":{"mode":"chat","base_model":"nous-hermes:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q3_K_S":{"mode":"chat","base_model":"nous-hermes:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q4_0":{"mode":"chat","base_model":"nous-hermes:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q4_1":{"mode":"chat","base_model":"nous-hermes:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q4_K_M":{"mode":"chat","base_model":"nous-hermes:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q4_K_S":{"mode":"chat","base_model":"nous-hermes:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q5_0":{"mode":"chat","base_model":"nous-hermes:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q5_1":{"mode":"chat","base_model":"nous-hermes:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q5_K_M":{"mode":"chat","base_model":"nous-hermes:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q5_K_S":{"mode":"chat","base_model":"nous-hermes:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q6_K":{"mode":"chat","base_model":"nous-hermes:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:13b-q8_0":{"mode":"chat","base_model":"nous-hermes:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-fp16":{"mode":"chat","base_model":"nous-hermes:70b-llama2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q2_K":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q3_K_L":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q3_K_M":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q3_K_S":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q4_0":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q4_1":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q4_K_M":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q4_K_S":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q5_0":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q5_1":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q5_K_M":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:70b-llama2-q6_K":{"mode":"chat","base_model":"nous-hermes:70b-llama2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2":{"mode":"chat","base_model":"nous-hermes:7b-llama2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-fp16":{"mode":"chat","base_model":"nous-hermes:7b-llama2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q2_K":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q3_K_L":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q3_K_M":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q3_K_S":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q4_0":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q4_1":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q4_K_M":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q4_K_S":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q5_0":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q5_1":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q5_K_M":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q5_K_S":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q6_K":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes:7b-llama2-q8_0":{"mode":"chat","base_model":"nous-hermes:7b-llama2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b":{"mode":"chat","base_model":"phind-codellama:34b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-fp16":{"mode":"chat","base_model":"phind-codellama:34b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python":{"mode":"chat","base_model":"phind-codellama:34b-python","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-fp16":{"mode":"chat","base_model":"phind-codellama:34b-python-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q2_K":{"mode":"chat","base_model":"phind-codellama:34b-python-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q3_K_L":{"mode":"chat","base_model":"phind-codellama:34b-python-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q3_K_M":{"mode":"chat","base_model":"phind-codellama:34b-python-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q3_K_S":{"mode":"chat","base_model":"phind-codellama:34b-python-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q4_0":{"mode":"chat","base_model":"phind-codellama:34b-python-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q4_1":{"mode":"chat","base_model":"phind-codellama:34b-python-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q4_K_M":{"mode":"chat","base_model":"phind-codellama:34b-python-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q4_K_S":{"mode":"chat","base_model":"phind-codellama:34b-python-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q5_0":{"mode":"chat","base_model":"phind-codellama:34b-python-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q5_1":{"mode":"chat","base_model":"phind-codellama:34b-python-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q5_K_M":{"mode":"chat","base_model":"phind-codellama:34b-python-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q5_K_S":{"mode":"chat","base_model":"phind-codellama:34b-python-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q6_K":{"mode":"chat","base_model":"phind-codellama:34b-python-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-python-q8_0":{"mode":"chat","base_model":"phind-codellama:34b-python-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q2_K":{"mode":"chat","base_model":"phind-codellama:34b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q3_K_L":{"mode":"chat","base_model":"phind-codellama:34b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q3_K_M":{"mode":"chat","base_model":"phind-codellama:34b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q3_K_S":{"mode":"chat","base_model":"phind-codellama:34b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q4_0":{"mode":"chat","base_model":"phind-codellama:34b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q4_1":{"mode":"chat","base_model":"phind-codellama:34b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q4_K_M":{"mode":"chat","base_model":"phind-codellama:34b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q4_K_S":{"mode":"chat","base_model":"phind-codellama:34b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q5_0":{"mode":"chat","base_model":"phind-codellama:34b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q5_1":{"mode":"chat","base_model":"phind-codellama:34b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q5_K_M":{"mode":"chat","base_model":"phind-codellama:34b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q5_K_S":{"mode":"chat","base_model":"phind-codellama:34b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q6_K":{"mode":"chat","base_model":"phind-codellama:34b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-q8_0":{"mode":"chat","base_model":"phind-codellama:34b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-fp16":{"mode":"chat","base_model":"phind-codellama:34b-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q2_K":{"mode":"chat","base_model":"phind-codellama:34b-v2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q3_K_L":{"mode":"chat","base_model":"phind-codellama:34b-v2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q3_K_M":{"mode":"chat","base_model":"phind-codellama:34b-v2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q3_K_S":{"mode":"chat","base_model":"phind-codellama:34b-v2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q4_0":{"mode":"chat","base_model":"phind-codellama:34b-v2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q4_1":{"mode":"chat","base_model":"phind-codellama:34b-v2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q4_K_M":{"mode":"chat","base_model":"phind-codellama:34b-v2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q4_K_S":{"mode":"chat","base_model":"phind-codellama:34b-v2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q5_0":{"mode":"chat","base_model":"phind-codellama:34b-v2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q5_1":{"mode":"chat","base_model":"phind-codellama:34b-v2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q5_K_M":{"mode":"chat","base_model":"phind-codellama:34b-v2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q5_K_S":{"mode":"chat","base_model":"phind-codellama:34b-v2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q6_K":{"mode":"chat","base_model":"phind-codellama:34b-v2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phind-codellama:34b-v2-q8_0":{"mode":"chat","base_model":"phind-codellama:34b-v2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava-phi3:3.8b":{"mode":"chat","base_model":"llava-phi3:3.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava-phi3:3.8b-mini-fp16":{"mode":"chat","base_model":"llava-phi3:3.8b-mini-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llava-phi3:3.8b-mini-q4_0":{"mode":"chat","base_model":"llava-phi3:3.8b-mini-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b":{"mode":"chat","base_model":"yarn-llama2:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b":{"mode":"chat","base_model":"yarn-llama2:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k":{"mode":"chat","base_model":"yarn-llama2:13b-128k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-fp16":{"mode":"chat","base_model":"yarn-llama2:13b-128k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q2_K":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q3_K_L":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q3_K_M":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q3_K_S":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q4_0":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q4_1":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q4_K_M":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q4_K_S":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q5_0":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q5_1":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q5_K_M":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q5_K_S":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q6_K":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-128k-q8_0":{"mode":"chat","base_model":"yarn-llama2:13b-128k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k":{"mode":"chat","base_model":"yarn-llama2:13b-64k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-fp16":{"mode":"chat","base_model":"yarn-llama2:13b-64k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q2_K":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q3_K_L":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q3_K_M":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q3_K_S":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q4_0":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q4_1":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q4_K_M":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q4_K_S":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q5_0":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q5_1":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q5_K_M":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q5_K_S":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q6_K":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:13b-64k-q8_0":{"mode":"chat","base_model":"yarn-llama2:13b-64k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k":{"mode":"chat","base_model":"yarn-llama2:7b-128k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-fp16":{"mode":"chat","base_model":"yarn-llama2:7b-128k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q2_K":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q3_K_L":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q3_K_M":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q3_K_S":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q4_0":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q4_1":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q4_K_M":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q4_K_S":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q5_0":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q5_1":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q5_K_M":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q5_K_S":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q6_K":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-128k-q8_0":{"mode":"chat","base_model":"yarn-llama2:7b-128k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k":{"mode":"chat","base_model":"yarn-llama2:7b-64k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-fp16":{"mode":"chat","base_model":"yarn-llama2:7b-64k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q2_K":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q3_K_L":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q3_K_M":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q3_K_S":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q4_0":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q4_1":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q4_K_M":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q4_K_S":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q5_0":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q5_1":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q5_K_M":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q5_K_S":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q6_K":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-llama2:7b-64k-q8_0":{"mode":"chat","base_model":"yarn-llama2:7b-64k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b":{"mode":"chat","base_model":"solar:10.7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-fp16":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q2_K":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q3_K_L":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q3_K_M":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q3_K_S":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q4_0":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q4_1":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q4_K_M":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q4_K_S":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q5_0":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q5_1":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q5_K_M":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q5_K_S":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q6_K":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-instruct-v1-q8_0":{"mode":"chat","base_model":"solar:10.7b-instruct-v1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-fp16":{"mode":"chat","base_model":"solar:10.7b-text-v1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q2_K":{"mode":"chat","base_model":"solar:10.7b-text-v1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q3_K_L":{"mode":"chat","base_model":"solar:10.7b-text-v1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q3_K_M":{"mode":"chat","base_model":"solar:10.7b-text-v1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q3_K_S":{"mode":"chat","base_model":"solar:10.7b-text-v1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q4_0":{"mode":"chat","base_model":"solar:10.7b-text-v1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q4_1":{"mode":"chat","base_model":"solar:10.7b-text-v1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q4_K_M":{"mode":"chat","base_model":"solar:10.7b-text-v1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q4_K_S":{"mode":"chat","base_model":"solar:10.7b-text-v1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q5_0":{"mode":"chat","base_model":"solar:10.7b-text-v1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q5_1":{"mode":"chat","base_model":"solar:10.7b-text-v1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q5_K_M":{"mode":"chat","base_model":"solar:10.7b-text-v1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q5_K_S":{"mode":"chat","base_model":"solar:10.7b-text-v1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q6_K":{"mode":"chat","base_model":"solar:10.7b-text-v1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar:10.7b-text-v1-q8_0":{"mode":"chat","base_model":"solar:10.7b-text-v1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b":{"mode":"chat","base_model":"granite3.1-dense:2b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b":{"mode":"chat","base_model":"granite3.1-dense:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-fp16":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q2_K":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q3_K_L":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q3_K_M":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q3_K_S":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q4_0":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q4_1":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q4_K_S":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q5_0":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q5_1":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q5_K_M":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q5_K_S":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q6_K":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:2b-instruct-q8_0":{"mode":"chat","base_model":"granite3.1-dense:2b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-fp16":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q2_K":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q3_K_L":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q3_K_M":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q3_K_S":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q4_0":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q4_1":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q4_K_S":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q5_0":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q5_1":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q5_K_M":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q5_K_S":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q6_K":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-dense:8b-instruct-q8_0":{"mode":"chat","base_model":"granite3.1-dense:8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b":{"mode":"chat","base_model":"starling-lm:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha":{"mode":"chat","base_model":"starling-lm:7b-alpha","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-fp16":{"mode":"chat","base_model":"starling-lm:7b-alpha-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q2_K":{"mode":"chat","base_model":"starling-lm:7b-alpha-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q3_K_L":{"mode":"chat","base_model":"starling-lm:7b-alpha-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q3_K_M":{"mode":"chat","base_model":"starling-lm:7b-alpha-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q3_K_S":{"mode":"chat","base_model":"starling-lm:7b-alpha-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q4_0":{"mode":"chat","base_model":"starling-lm:7b-alpha-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q4_1":{"mode":"chat","base_model":"starling-lm:7b-alpha-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q4_K_M":{"mode":"chat","base_model":"starling-lm:7b-alpha-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q4_K_S":{"mode":"chat","base_model":"starling-lm:7b-alpha-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q5_0":{"mode":"chat","base_model":"starling-lm:7b-alpha-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q5_1":{"mode":"chat","base_model":"starling-lm:7b-alpha-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q5_K_M":{"mode":"chat","base_model":"starling-lm:7b-alpha-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q5_K_S":{"mode":"chat","base_model":"starling-lm:7b-alpha-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q6_K":{"mode":"chat","base_model":"starling-lm:7b-alpha-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-alpha-q8_0":{"mode":"chat","base_model":"starling-lm:7b-alpha-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta":{"mode":"chat","base_model":"starling-lm:7b-beta","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-fp16":{"mode":"chat","base_model":"starling-lm:7b-beta-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q2_K":{"mode":"chat","base_model":"starling-lm:7b-beta-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q3_K_L":{"mode":"chat","base_model":"starling-lm:7b-beta-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q3_K_M":{"mode":"chat","base_model":"starling-lm:7b-beta-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q3_K_S":{"mode":"chat","base_model":"starling-lm:7b-beta-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q4_0":{"mode":"chat","base_model":"starling-lm:7b-beta-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q4_1":{"mode":"chat","base_model":"starling-lm:7b-beta-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q4_K_M":{"mode":"chat","base_model":"starling-lm:7b-beta-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q4_K_S":{"mode":"chat","base_model":"starling-lm:7b-beta-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q5_0":{"mode":"chat","base_model":"starling-lm:7b-beta-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q5_1":{"mode":"chat","base_model":"starling-lm:7b-beta-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q5_K_M":{"mode":"chat","base_model":"starling-lm:7b-beta-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q5_K_S":{"mode":"chat","base_model":"starling-lm:7b-beta-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q6_K":{"mode":"chat","base_model":"starling-lm:7b-beta-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:7b-beta-q8_0":{"mode":"chat","base_model":"starling-lm:7b-beta-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:alpha":{"mode":"chat","base_model":"starling-lm:alpha","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"starling-lm:beta":{"mode":"chat","base_model":"starling-lm:beta","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b":{"mode":"chat","base_model":"athene-v2:72b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-fp16":{"mode":"chat","base_model":"athene-v2:72b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q2_K":{"mode":"chat","base_model":"athene-v2:72b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q3_K_L":{"mode":"chat","base_model":"athene-v2:72b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q3_K_M":{"mode":"chat","base_model":"athene-v2:72b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q3_K_S":{"mode":"chat","base_model":"athene-v2:72b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q4_0":{"mode":"chat","base_model":"athene-v2:72b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q4_1":{"mode":"chat","base_model":"athene-v2:72b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q4_K_M":{"mode":"chat","base_model":"athene-v2:72b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q4_K_S":{"mode":"chat","base_model":"athene-v2:72b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q5_0":{"mode":"chat","base_model":"athene-v2:72b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q5_1":{"mode":"chat","base_model":"athene-v2:72b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q5_K_M":{"mode":"chat","base_model":"athene-v2:72b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q5_K_S":{"mode":"chat","base_model":"athene-v2:72b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q6_K":{"mode":"chat","base_model":"athene-v2:72b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"athene-v2:72b-q8_0":{"mode":"chat","base_model":"athene-v2:72b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b":{"mode":"chat","base_model":"yi-coder:1.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b":{"mode":"chat","base_model":"yi-coder:9b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base":{"mode":"chat","base_model":"yi-coder:1.5b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-fp16":{"mode":"chat","base_model":"yi-coder:1.5b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q2_K":{"mode":"chat","base_model":"yi-coder:1.5b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q3_K_L":{"mode":"chat","base_model":"yi-coder:1.5b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q3_K_M":{"mode":"chat","base_model":"yi-coder:1.5b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q3_K_S":{"mode":"chat","base_model":"yi-coder:1.5b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q4_0":{"mode":"chat","base_model":"yi-coder:1.5b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q4_1":{"mode":"chat","base_model":"yi-coder:1.5b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q4_K_M":{"mode":"chat","base_model":"yi-coder:1.5b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q4_K_S":{"mode":"chat","base_model":"yi-coder:1.5b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q5_0":{"mode":"chat","base_model":"yi-coder:1.5b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q5_1":{"mode":"chat","base_model":"yi-coder:1.5b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q5_K_M":{"mode":"chat","base_model":"yi-coder:1.5b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q5_K_S":{"mode":"chat","base_model":"yi-coder:1.5b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q6_K":{"mode":"chat","base_model":"yi-coder:1.5b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-base-q8_0":{"mode":"chat","base_model":"yi-coder:1.5b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat":{"mode":"chat","base_model":"yi-coder:1.5b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-fp16":{"mode":"chat","base_model":"yi-coder:1.5b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q2_K":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q3_K_L":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q3_K_M":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q3_K_S":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q4_0":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q4_1":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q4_K_M":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q4_K_S":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q5_0":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q5_1":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q5_K_M":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q5_K_S":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q6_K":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:1.5b-chat-q8_0":{"mode":"chat","base_model":"yi-coder:1.5b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base":{"mode":"chat","base_model":"yi-coder:9b-base","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-fp16":{"mode":"chat","base_model":"yi-coder:9b-base-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q2_K":{"mode":"chat","base_model":"yi-coder:9b-base-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q3_K_L":{"mode":"chat","base_model":"yi-coder:9b-base-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q3_K_M":{"mode":"chat","base_model":"yi-coder:9b-base-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q3_K_S":{"mode":"chat","base_model":"yi-coder:9b-base-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q4_0":{"mode":"chat","base_model":"yi-coder:9b-base-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q4_1":{"mode":"chat","base_model":"yi-coder:9b-base-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q4_K_M":{"mode":"chat","base_model":"yi-coder:9b-base-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q4_K_S":{"mode":"chat","base_model":"yi-coder:9b-base-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q5_0":{"mode":"chat","base_model":"yi-coder:9b-base-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q5_1":{"mode":"chat","base_model":"yi-coder:9b-base-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q5_K_M":{"mode":"chat","base_model":"yi-coder:9b-base-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q5_K_S":{"mode":"chat","base_model":"yi-coder:9b-base-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q6_K":{"mode":"chat","base_model":"yi-coder:9b-base-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-base-q8_0":{"mode":"chat","base_model":"yi-coder:9b-base-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat":{"mode":"chat","base_model":"yi-coder:9b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-fp16":{"mode":"chat","base_model":"yi-coder:9b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q2_K":{"mode":"chat","base_model":"yi-coder:9b-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q3_K_L":{"mode":"chat","base_model":"yi-coder:9b-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q3_K_M":{"mode":"chat","base_model":"yi-coder:9b-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q3_K_S":{"mode":"chat","base_model":"yi-coder:9b-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q4_0":{"mode":"chat","base_model":"yi-coder:9b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q4_1":{"mode":"chat","base_model":"yi-coder:9b-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q4_K_M":{"mode":"chat","base_model":"yi-coder:9b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q4_K_S":{"mode":"chat","base_model":"yi-coder:9b-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q5_0":{"mode":"chat","base_model":"yi-coder:9b-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q5_1":{"mode":"chat","base_model":"yi-coder:9b-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q5_K_M":{"mode":"chat","base_model":"yi-coder:9b-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q5_K_S":{"mode":"chat","base_model":"yi-coder:9b-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q6_K":{"mode":"chat","base_model":"yi-coder:9b-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yi-coder:9b-chat-q8_0":{"mode":"chat","base_model":"yi-coder:9b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-fp16":{"mode":"chat","base_model":"wizardlm:13b-llama2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q2_K":{"mode":"chat","base_model":"wizardlm:13b-llama2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q3_K_L":{"mode":"chat","base_model":"wizardlm:13b-llama2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q3_K_M":{"mode":"chat","base_model":"wizardlm:13b-llama2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q3_K_S":{"mode":"chat","base_model":"wizardlm:13b-llama2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q4_0":{"mode":"chat","base_model":"wizardlm:13b-llama2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q4_1":{"mode":"chat","base_model":"wizardlm:13b-llama2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q4_K_M":{"mode":"chat","base_model":"wizardlm:13b-llama2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q4_K_S":{"mode":"chat","base_model":"wizardlm:13b-llama2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q5_0":{"mode":"chat","base_model":"wizardlm:13b-llama2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q5_1":{"mode":"chat","base_model":"wizardlm:13b-llama2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q5_K_M":{"mode":"chat","base_model":"wizardlm:13b-llama2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q5_K_S":{"mode":"chat","base_model":"wizardlm:13b-llama2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q6_K":{"mode":"chat","base_model":"wizardlm:13b-llama2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-llama2-q8_0":{"mode":"chat","base_model":"wizardlm:13b-llama2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q2_K":{"mode":"chat","base_model":"wizardlm:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q3_K_L":{"mode":"chat","base_model":"wizardlm:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q3_K_M":{"mode":"chat","base_model":"wizardlm:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q3_K_S":{"mode":"chat","base_model":"wizardlm:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q4_0":{"mode":"chat","base_model":"wizardlm:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q4_1":{"mode":"chat","base_model":"wizardlm:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q4_K_M":{"mode":"chat","base_model":"wizardlm:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q4_K_S":{"mode":"chat","base_model":"wizardlm:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q5_0":{"mode":"chat","base_model":"wizardlm:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q5_1":{"mode":"chat","base_model":"wizardlm:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q5_K_M":{"mode":"chat","base_model":"wizardlm:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q5_K_S":{"mode":"chat","base_model":"wizardlm:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q6_K":{"mode":"chat","base_model":"wizardlm:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:13b-q8_0":{"mode":"chat","base_model":"wizardlm:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-fp16":{"mode":"chat","base_model":"wizardlm:30b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q2_K":{"mode":"chat","base_model":"wizardlm:30b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q3_K_L":{"mode":"chat","base_model":"wizardlm:30b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q3_K_M":{"mode":"chat","base_model":"wizardlm:30b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q3_K_S":{"mode":"chat","base_model":"wizardlm:30b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q4_0":{"mode":"chat","base_model":"wizardlm:30b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q4_1":{"mode":"chat","base_model":"wizardlm:30b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q4_K_M":{"mode":"chat","base_model":"wizardlm:30b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q4_K_S":{"mode":"chat","base_model":"wizardlm:30b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q5_0":{"mode":"chat","base_model":"wizardlm:30b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q5_1":{"mode":"chat","base_model":"wizardlm:30b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q5_K_M":{"mode":"chat","base_model":"wizardlm:30b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q5_K_S":{"mode":"chat","base_model":"wizardlm:30b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q6_K":{"mode":"chat","base_model":"wizardlm:30b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:30b-q8_0":{"mode":"chat","base_model":"wizardlm:30b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q2_K":{"mode":"chat","base_model":"wizardlm:70b-llama2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q3_K_L":{"mode":"chat","base_model":"wizardlm:70b-llama2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q3_K_M":{"mode":"chat","base_model":"wizardlm:70b-llama2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q3_K_S":{"mode":"chat","base_model":"wizardlm:70b-llama2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q4_0":{"mode":"chat","base_model":"wizardlm:70b-llama2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q4_1":{"mode":"chat","base_model":"wizardlm:70b-llama2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q4_K_M":{"mode":"chat","base_model":"wizardlm:70b-llama2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q4_K_S":{"mode":"chat","base_model":"wizardlm:70b-llama2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q5_0":{"mode":"chat","base_model":"wizardlm:70b-llama2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q5_K_M":{"mode":"chat","base_model":"wizardlm:70b-llama2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q5_K_S":{"mode":"chat","base_model":"wizardlm:70b-llama2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q6_K":{"mode":"chat","base_model":"wizardlm:70b-llama2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:70b-llama2-q8_0":{"mode":"chat","base_model":"wizardlm:70b-llama2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-fp16":{"mode":"chat","base_model":"wizardlm:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q2_K":{"mode":"chat","base_model":"wizardlm:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q3_K_L":{"mode":"chat","base_model":"wizardlm:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q3_K_M":{"mode":"chat","base_model":"wizardlm:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q3_K_S":{"mode":"chat","base_model":"wizardlm:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q4_0":{"mode":"chat","base_model":"wizardlm:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q4_1":{"mode":"chat","base_model":"wizardlm:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q4_K_M":{"mode":"chat","base_model":"wizardlm:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q4_K_S":{"mode":"chat","base_model":"wizardlm:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q5_0":{"mode":"chat","base_model":"wizardlm:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q5_1":{"mode":"chat","base_model":"wizardlm:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q5_K_M":{"mode":"chat","base_model":"wizardlm:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q5_K_S":{"mode":"chat","base_model":"wizardlm:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q6_K":{"mode":"chat","base_model":"wizardlm:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm:7b-q8_0":{"mode":"chat","base_model":"wizardlm:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b":{"mode":"chat","base_model":"samantha-mistral:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-fp16":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q2_K":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q3_K_L":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q3_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q3_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q4_0":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q4_1":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q4_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q4_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q5_0":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q5_1":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q5_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q5_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q6_K":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-instruct-q8_0":{"mode":"chat","base_model":"samantha-mistral:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text":{"mode":"chat","base_model":"samantha-mistral:7b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-fp16":{"mode":"chat","base_model":"samantha-mistral:7b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q2_K":{"mode":"chat","base_model":"samantha-mistral:7b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q3_K_L":{"mode":"chat","base_model":"samantha-mistral:7b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q3_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q3_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q4_0":{"mode":"chat","base_model":"samantha-mistral:7b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q4_1":{"mode":"chat","base_model":"samantha-mistral:7b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q4_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q4_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q5_0":{"mode":"chat","base_model":"samantha-mistral:7b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q5_1":{"mode":"chat","base_model":"samantha-mistral:7b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q5_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q5_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q6_K":{"mode":"chat","base_model":"samantha-mistral:7b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-text-q8_0":{"mode":"chat","base_model":"samantha-mistral:7b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-fp16":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q2_K":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q3_K_L":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q3_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q3_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q4_0":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q4_1":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q4_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q4_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q5_0":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q5_1":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q5_K_M":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q5_K_S":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q6_K":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"samantha-mistral:7b-v1.2-text-q8_0":{"mode":"chat","base_model":"samantha-mistral:7b-v1.2-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1m":{"mode":"chat","base_model":"internlm2:1m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b":{"mode":"chat","base_model":"internlm2:1.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b":{"mode":"chat","base_model":"internlm2:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b":{"mode":"chat","base_model":"internlm2:20b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-fp16":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q2_K":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q3_K_L":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q3_K_M":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q3_K_S":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q4_0":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q4_1":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q4_K_M":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q4_K_S":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q5_0":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q5_1":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q5_K_M":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q5_K_S":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q6_K":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:1.8b-chat-v2.5-q8_0":{"mode":"chat","base_model":"internlm2:1.8b-chat-v2.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-fp16":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q2_K":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q3_K_L":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q3_K_M":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q3_K_S":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q4_0":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q4_1":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q4_K_M":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q4_K_S":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q5_0":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q5_1":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q5_K_M":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q5_K_S":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q6_K":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:20b-chat-v2.5-q8_0":{"mode":"chat","base_model":"internlm2:20b-chat-v2.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-fp16":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q2_K":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q3_K_L":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q3_K_M":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q3_K_S":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q4_0":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q4_1":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q4_K_M":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q4_K_S":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q5_0":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q5_1":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q5_K_M":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q5_K_S":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q6_K":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-1m-v2.5-q8_0":{"mode":"chat","base_model":"internlm2:7b-chat-1m-v2.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-fp16":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q2_K":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q3_K_L":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q3_K_M":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q3_K_S":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q4_0":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q4_1":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q4_K_M":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q4_K_S":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q5_0":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q5_1":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q5_K_M":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q5_K_S":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q6_K":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"internlm2:7b-chat-v2.5-q8_0":{"mode":"chat","base_model":"internlm2:7b-chat-v2.5-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b":{"mode":"chat","base_model":"falcon:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b":{"mode":"chat","base_model":"falcon:40b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:180b":{"mode":"chat","base_model":"falcon:180b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:180b-chat":{"mode":"chat","base_model":"falcon:180b-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:180b-chat-q4_0":{"mode":"chat","base_model":"falcon:180b-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:180b-text":{"mode":"chat","base_model":"falcon:180b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:180b-text-q4_0":{"mode":"chat","base_model":"falcon:180b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-instruct":{"mode":"chat","base_model":"falcon:40b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-instruct-fp16":{"mode":"chat","base_model":"falcon:40b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-instruct-q4_0":{"mode":"chat","base_model":"falcon:40b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-instruct-q4_1":{"mode":"chat","base_model":"falcon:40b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-instruct-q5_0":{"mode":"chat","base_model":"falcon:40b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-instruct-q5_1":{"mode":"chat","base_model":"falcon:40b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-instruct-q8_0":{"mode":"chat","base_model":"falcon:40b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-text":{"mode":"chat","base_model":"falcon:40b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-text-fp16":{"mode":"chat","base_model":"falcon:40b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-text-q4_0":{"mode":"chat","base_model":"falcon:40b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-text-q4_1":{"mode":"chat","base_model":"falcon:40b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-text-q5_0":{"mode":"chat","base_model":"falcon:40b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-text-q5_1":{"mode":"chat","base_model":"falcon:40b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:40b-text-q8_0":{"mode":"chat","base_model":"falcon:40b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-instruct":{"mode":"chat","base_model":"falcon:7b-instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-instruct-fp16":{"mode":"chat","base_model":"falcon:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-instruct-q4_0":{"mode":"chat","base_model":"falcon:7b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-instruct-q4_1":{"mode":"chat","base_model":"falcon:7b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-instruct-q5_0":{"mode":"chat","base_model":"falcon:7b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-instruct-q5_1":{"mode":"chat","base_model":"falcon:7b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-instruct-q8_0":{"mode":"chat","base_model":"falcon:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-text":{"mode":"chat","base_model":"falcon:7b-text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-text-fp16":{"mode":"chat","base_model":"falcon:7b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-text-q4_0":{"mode":"chat","base_model":"falcon:7b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-text-q4_1":{"mode":"chat","base_model":"falcon:7b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-text-q5_0":{"mode":"chat","base_model":"falcon:7b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-text-q5_1":{"mode":"chat","base_model":"falcon:7b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:7b-text-q8_0":{"mode":"chat","base_model":"falcon:7b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:instruct":{"mode":"chat","base_model":"falcon:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon:text":{"mode":"chat","base_model":"falcon:text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b":{"mode":"chat","base_model":"nemotron-mini:4b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-fp16":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q2_K":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q3_K_L":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q3_K_M":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q3_K_S":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q4_0":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q4_1":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q4_K_M":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q4_K_S":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q5_0":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q5_1":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q5_K_M":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q5_K_S":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q6_K":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron-mini:4b-instruct-q8_0":{"mode":"chat","base_model":"nemotron-mini:4b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b":{"mode":"chat","base_model":"nemotron:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-fp16":{"mode":"chat","base_model":"nemotron:70b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q2_K":{"mode":"chat","base_model":"nemotron:70b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q3_K_L":{"mode":"chat","base_model":"nemotron:70b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q3_K_M":{"mode":"chat","base_model":"nemotron:70b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q3_K_S":{"mode":"chat","base_model":"nemotron:70b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q4_0":{"mode":"chat","base_model":"nemotron:70b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q4_1":{"mode":"chat","base_model":"nemotron:70b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q4_K_M":{"mode":"chat","base_model":"nemotron:70b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q4_K_S":{"mode":"chat","base_model":"nemotron:70b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q5_0":{"mode":"chat","base_model":"nemotron:70b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q5_1":{"mode":"chat","base_model":"nemotron:70b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q5_K_M":{"mode":"chat","base_model":"nemotron:70b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q5_K_S":{"mode":"chat","base_model":"nemotron:70b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q6_K":{"mode":"chat","base_model":"nemotron:70b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nemotron:70b-instruct-q8_0":{"mode":"chat","base_model":"nemotron:70b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b":{"mode":"chat","base_model":"dolphin-phi:2.7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q2_K":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q3_K_L":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q3_K_M":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q3_K_S":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q4_0":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q4_K_M":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q4_K_S":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q5_0":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q5_K_M":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q5_K_S":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q6_K":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dolphin-phi:2.7b-v2.6-q8_0":{"mode":"chat","base_model":"dolphin-phi:2.7b-v2.6-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b":{"mode":"chat","base_model":"orca2:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b":{"mode":"chat","base_model":"orca2:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-fp16":{"mode":"chat","base_model":"orca2:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q2_K":{"mode":"chat","base_model":"orca2:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q3_K_L":{"mode":"chat","base_model":"orca2:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q3_K_M":{"mode":"chat","base_model":"orca2:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q3_K_S":{"mode":"chat","base_model":"orca2:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q4_0":{"mode":"chat","base_model":"orca2:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q4_1":{"mode":"chat","base_model":"orca2:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q4_K_M":{"mode":"chat","base_model":"orca2:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q4_K_S":{"mode":"chat","base_model":"orca2:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q5_0":{"mode":"chat","base_model":"orca2:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q5_1":{"mode":"chat","base_model":"orca2:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q5_K_M":{"mode":"chat","base_model":"orca2:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q5_K_S":{"mode":"chat","base_model":"orca2:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q6_K":{"mode":"chat","base_model":"orca2:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:13b-q8_0":{"mode":"chat","base_model":"orca2:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-fp16":{"mode":"chat","base_model":"orca2:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q2_K":{"mode":"chat","base_model":"orca2:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q3_K_L":{"mode":"chat","base_model":"orca2:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q3_K_M":{"mode":"chat","base_model":"orca2:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q3_K_S":{"mode":"chat","base_model":"orca2:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q4_0":{"mode":"chat","base_model":"orca2:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q4_1":{"mode":"chat","base_model":"orca2:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q4_K_M":{"mode":"chat","base_model":"orca2:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q4_K_S":{"mode":"chat","base_model":"orca2:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q5_0":{"mode":"chat","base_model":"orca2:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q5_1":{"mode":"chat","base_model":"orca2:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q5_K_M":{"mode":"chat","base_model":"orca2:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q5_K_S":{"mode":"chat","base_model":"orca2:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q6_K":{"mode":"chat","base_model":"orca2:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"orca2:7b-q8_0":{"mode":"chat","base_model":"orca2:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepscaler:1.5b":{"mode":"chat","base_model":"deepscaler:1.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepscaler:1.5b-preview-fp16":{"mode":"chat","base_model":"deepscaler:1.5b-preview-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepscaler:1.5b-preview-q4_K_M":{"mode":"chat","base_model":"deepscaler:1.5b-preview-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepscaler:1.5b-preview-q8_0":{"mode":"chat","base_model":"deepscaler:1.5b-preview-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b":{"mode":"chat","base_model":"wizardlm-uncensored:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-fp16":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q2_K":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q3_K_L":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q3_K_M":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q3_K_S":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q4_0":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q4_1":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q4_K_M":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q4_K_S":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q5_0":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q5_1":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q5_K_M":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q5_K_S":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q6_K":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizardlm-uncensored:13b-llama2-q8_0":{"mode":"chat","base_model":"wizardlm-uncensored:13b-llama2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b":{"mode":"chat","base_model":"stable-beluga:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b":{"mode":"chat","base_model":"stable-beluga:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b":{"mode":"chat","base_model":"stable-beluga:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-fp16":{"mode":"chat","base_model":"stable-beluga:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q2_K":{"mode":"chat","base_model":"stable-beluga:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q3_K_L":{"mode":"chat","base_model":"stable-beluga:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q3_K_M":{"mode":"chat","base_model":"stable-beluga:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q3_K_S":{"mode":"chat","base_model":"stable-beluga:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q4_0":{"mode":"chat","base_model":"stable-beluga:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q4_1":{"mode":"chat","base_model":"stable-beluga:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q4_K_M":{"mode":"chat","base_model":"stable-beluga:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q4_K_S":{"mode":"chat","base_model":"stable-beluga:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q5_0":{"mode":"chat","base_model":"stable-beluga:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q5_1":{"mode":"chat","base_model":"stable-beluga:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q5_K_M":{"mode":"chat","base_model":"stable-beluga:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q5_K_S":{"mode":"chat","base_model":"stable-beluga:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q6_K":{"mode":"chat","base_model":"stable-beluga:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:13b-q8_0":{"mode":"chat","base_model":"stable-beluga:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-fp16":{"mode":"chat","base_model":"stable-beluga:70b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q2_K":{"mode":"chat","base_model":"stable-beluga:70b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q3_K_L":{"mode":"chat","base_model":"stable-beluga:70b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q3_K_M":{"mode":"chat","base_model":"stable-beluga:70b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q3_K_S":{"mode":"chat","base_model":"stable-beluga:70b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q4_0":{"mode":"chat","base_model":"stable-beluga:70b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q4_1":{"mode":"chat","base_model":"stable-beluga:70b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q4_K_M":{"mode":"chat","base_model":"stable-beluga:70b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q4_K_S":{"mode":"chat","base_model":"stable-beluga:70b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q5_0":{"mode":"chat","base_model":"stable-beluga:70b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q5_1":{"mode":"chat","base_model":"stable-beluga:70b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q5_K_M":{"mode":"chat","base_model":"stable-beluga:70b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q5_K_S":{"mode":"chat","base_model":"stable-beluga:70b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q6_K":{"mode":"chat","base_model":"stable-beluga:70b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:70b-q8_0":{"mode":"chat","base_model":"stable-beluga:70b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-fp16":{"mode":"chat","base_model":"stable-beluga:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q2_K":{"mode":"chat","base_model":"stable-beluga:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q3_K_L":{"mode":"chat","base_model":"stable-beluga:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q3_K_M":{"mode":"chat","base_model":"stable-beluga:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q3_K_S":{"mode":"chat","base_model":"stable-beluga:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q4_0":{"mode":"chat","base_model":"stable-beluga:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q4_1":{"mode":"chat","base_model":"stable-beluga:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q4_K_M":{"mode":"chat","base_model":"stable-beluga:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q4_K_S":{"mode":"chat","base_model":"stable-beluga:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q5_0":{"mode":"chat","base_model":"stable-beluga:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q5_1":{"mode":"chat","base_model":"stable-beluga:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q5_K_M":{"mode":"chat","base_model":"stable-beluga:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q5_K_S":{"mode":"chat","base_model":"stable-beluga:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q6_K":{"mode":"chat","base_model":"stable-beluga:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stable-beluga:7b-q8_0":{"mode":"chat","base_model":"stable-beluga:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b":{"mode":"chat","base_model":"granite3-dense:2b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b":{"mode":"chat","base_model":"granite3-dense:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-fp16":{"mode":"chat","base_model":"granite3-dense:2b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q2_K":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q3_K_L":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q3_K_M":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q3_K_S":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q4_0":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q4_1":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q4_K_S":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q5_0":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q5_1":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q5_K_M":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q5_K_S":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q6_K":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:2b-instruct-q8_0":{"mode":"chat","base_model":"granite3-dense:2b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-fp16":{"mode":"chat","base_model":"granite3-dense:8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q2_K":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q3_K_L":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q3_K_M":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q3_K_S":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q4_0":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q4_1":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q4_K_S":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q5_0":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q5_1":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q5_K_M":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q5_K_S":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q6_K":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-dense:8b-instruct-q8_0":{"mode":"chat","base_model":"granite3-dense:8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b":{"mode":"chat","base_model":"llama3-groq-tool-use:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b":{"mode":"chat","base_model":"llama3-groq-tool-use:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-fp16":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q2_K":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q3_K_L":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q3_K_M":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q3_K_S":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q4_0":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q4_1":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q4_K_M":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q4_K_S":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q5_0":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q5_1":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q5_K_M":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q5_K_S":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q6_K":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:70b-q8_0":{"mode":"chat","base_model":"llama3-groq-tool-use:70b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-fp16":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q2_K":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q3_K_L":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q3_K_M":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q3_K_S":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q4_0":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q4_1":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q4_K_M":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q4_K_S":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q5_0":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q5_1":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q5_K_M":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q5_K_S":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q6_K":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama3-groq-tool-use:8b-q8_0":{"mode":"chat","base_model":"llama3-groq-tool-use:8b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2.5:236b":{"mode":"chat","base_model":"deepseek-v2.5:236b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2.5:236b-q4_0":{"mode":"chat","base_model":"deepseek-v2.5:236b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2.5:236b-q4_1":{"mode":"chat","base_model":"deepseek-v2.5:236b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2.5:236b-q5_0":{"mode":"chat","base_model":"deepseek-v2.5:236b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2.5:236b-q5_1":{"mode":"chat","base_model":"deepseek-v2.5:236b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"deepseek-v2.5:236b-q8_0":{"mode":"chat","base_model":"deepseek-v2.5:236b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b":{"mode":"chat","base_model":"medllama2:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-fp16":{"mode":"chat","base_model":"medllama2:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q2_K":{"mode":"chat","base_model":"medllama2:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q3_K_L":{"mode":"chat","base_model":"medllama2:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q3_K_M":{"mode":"chat","base_model":"medllama2:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q3_K_S":{"mode":"chat","base_model":"medllama2:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q4_0":{"mode":"chat","base_model":"medllama2:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q4_1":{"mode":"chat","base_model":"medllama2:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q4_K_M":{"mode":"chat","base_model":"medllama2:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q4_K_S":{"mode":"chat","base_model":"medllama2:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q5_0":{"mode":"chat","base_model":"medllama2:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q5_1":{"mode":"chat","base_model":"medllama2:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q5_K_M":{"mode":"chat","base_model":"medllama2:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q5_K_S":{"mode":"chat","base_model":"medllama2:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q6_K":{"mode":"chat","base_model":"medllama2:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"medllama2:7b-q8_0":{"mode":"chat","base_model":"medllama2:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b":{"mode":"chat","base_model":"meditron:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:70b":{"mode":"chat","base_model":"meditron:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:70b-q4_0":{"mode":"chat","base_model":"meditron:70b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:70b-q4_1":{"mode":"chat","base_model":"meditron:70b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:70b-q4_K_S":{"mode":"chat","base_model":"meditron:70b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:70b-q5_1":{"mode":"chat","base_model":"meditron:70b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-fp16":{"mode":"chat","base_model":"meditron:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q2_K":{"mode":"chat","base_model":"meditron:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q3_K_L":{"mode":"chat","base_model":"meditron:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q3_K_M":{"mode":"chat","base_model":"meditron:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q3_K_S":{"mode":"chat","base_model":"meditron:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q4_0":{"mode":"chat","base_model":"meditron:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q4_1":{"mode":"chat","base_model":"meditron:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q4_K_M":{"mode":"chat","base_model":"meditron:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q4_K_S":{"mode":"chat","base_model":"meditron:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q5_0":{"mode":"chat","base_model":"meditron:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q5_1":{"mode":"chat","base_model":"meditron:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q5_K_M":{"mode":"chat","base_model":"meditron:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q5_K_S":{"mode":"chat","base_model":"meditron:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q6_K":{"mode":"chat","base_model":"meditron:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"meditron:7b-q8_0":{"mode":"chat","base_model":"meditron:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smallthinker:3b":{"mode":"chat","base_model":"smallthinker:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smallthinker:3b-preview-fp16":{"mode":"chat","base_model":"smallthinker:3b-preview-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smallthinker:3b-preview-q4_K_M":{"mode":"chat","base_model":"smallthinker:3b-preview-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"smallthinker:3b-preview-q8_0":{"mode":"chat","base_model":"smallthinker:3b-preview-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"paraphrase-multilingual:278m":{"mode":"chat","base_model":"paraphrase-multilingual:278m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"paraphrase-multilingual:278m-mpnet-base-v2-fp16":{"mode":"chat","base_model":"paraphrase-multilingual:278m-mpnet-base-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-fp16":{"mode":"chat","base_model":"llama-pro:8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q2_K":{"mode":"chat","base_model":"llama-pro:8b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q3_K_L":{"mode":"chat","base_model":"llama-pro:8b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q3_K_M":{"mode":"chat","base_model":"llama-pro:8b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q3_K_S":{"mode":"chat","base_model":"llama-pro:8b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q4_0":{"mode":"chat","base_model":"llama-pro:8b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q4_1":{"mode":"chat","base_model":"llama-pro:8b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q4_K_M":{"mode":"chat","base_model":"llama-pro:8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q4_K_S":{"mode":"chat","base_model":"llama-pro:8b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q5_0":{"mode":"chat","base_model":"llama-pro:8b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q5_1":{"mode":"chat","base_model":"llama-pro:8b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q5_K_M":{"mode":"chat","base_model":"llama-pro:8b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q5_K_S":{"mode":"chat","base_model":"llama-pro:8b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q6_K":{"mode":"chat","base_model":"llama-pro:8b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-instruct-q8_0":{"mode":"chat","base_model":"llama-pro:8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-fp16":{"mode":"chat","base_model":"llama-pro:8b-text-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q2_K":{"mode":"chat","base_model":"llama-pro:8b-text-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q3_K_L":{"mode":"chat","base_model":"llama-pro:8b-text-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q3_K_M":{"mode":"chat","base_model":"llama-pro:8b-text-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q3_K_S":{"mode":"chat","base_model":"llama-pro:8b-text-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q4_0":{"mode":"chat","base_model":"llama-pro:8b-text-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q4_1":{"mode":"chat","base_model":"llama-pro:8b-text-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q4_K_M":{"mode":"chat","base_model":"llama-pro:8b-text-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q4_K_S":{"mode":"chat","base_model":"llama-pro:8b-text-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q5_0":{"mode":"chat","base_model":"llama-pro:8b-text-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q5_1":{"mode":"chat","base_model":"llama-pro:8b-text-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q5_K_M":{"mode":"chat","base_model":"llama-pro:8b-text-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q5_K_S":{"mode":"chat","base_model":"llama-pro:8b-text-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q6_K":{"mode":"chat","base_model":"llama-pro:8b-text-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:8b-text-q8_0":{"mode":"chat","base_model":"llama-pro:8b-text-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:instruct":{"mode":"chat","base_model":"llama-pro:instruct","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-pro:text":{"mode":"chat","base_model":"llama-pro:text","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b":{"mode":"chat","base_model":"yarn-mistral:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k":{"mode":"chat","base_model":"yarn-mistral:7b-128k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-fp16":{"mode":"chat","base_model":"yarn-mistral:7b-128k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q2_K":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q3_K_L":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q3_K_M":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q3_K_S":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q4_0":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q4_1":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q4_K_M":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q4_K_S":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q5_0":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q5_1":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q5_K_M":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q5_K_S":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q6_K":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-128k-q8_0":{"mode":"chat","base_model":"yarn-mistral:7b-128k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k":{"mode":"chat","base_model":"yarn-mistral:7b-64k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q2_K":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q3_K_L":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q3_K_M":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q3_K_S":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q4_0":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q4_1":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q4_K_M":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q4_K_S":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q5_0":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q5_1":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q5_K_M":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q5_K_S":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q6_K":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"yarn-mistral:7b-64k-q8_0":{"mode":"chat","base_model":"yarn-mistral:7b-64k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b":{"mode":"chat","base_model":"aya-expanse:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b":{"mode":"chat","base_model":"aya-expanse:32b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-fp16":{"mode":"chat","base_model":"aya-expanse:32b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q2_K":{"mode":"chat","base_model":"aya-expanse:32b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q3_K_L":{"mode":"chat","base_model":"aya-expanse:32b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q3_K_M":{"mode":"chat","base_model":"aya-expanse:32b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q3_K_S":{"mode":"chat","base_model":"aya-expanse:32b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q4_0":{"mode":"chat","base_model":"aya-expanse:32b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q4_1":{"mode":"chat","base_model":"aya-expanse:32b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q4_K_M":{"mode":"chat","base_model":"aya-expanse:32b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q4_K_S":{"mode":"chat","base_model":"aya-expanse:32b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q5_0":{"mode":"chat","base_model":"aya-expanse:32b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q5_1":{"mode":"chat","base_model":"aya-expanse:32b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q5_K_M":{"mode":"chat","base_model":"aya-expanse:32b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q5_K_S":{"mode":"chat","base_model":"aya-expanse:32b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q6_K":{"mode":"chat","base_model":"aya-expanse:32b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:32b-q8_0":{"mode":"chat","base_model":"aya-expanse:32b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-fp16":{"mode":"chat","base_model":"aya-expanse:8b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q2_K":{"mode":"chat","base_model":"aya-expanse:8b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q3_K_L":{"mode":"chat","base_model":"aya-expanse:8b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q3_K_M":{"mode":"chat","base_model":"aya-expanse:8b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q3_K_S":{"mode":"chat","base_model":"aya-expanse:8b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q4_0":{"mode":"chat","base_model":"aya-expanse:8b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q4_1":{"mode":"chat","base_model":"aya-expanse:8b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q4_K_M":{"mode":"chat","base_model":"aya-expanse:8b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q4_K_S":{"mode":"chat","base_model":"aya-expanse:8b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q5_0":{"mode":"chat","base_model":"aya-expanse:8b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q5_1":{"mode":"chat","base_model":"aya-expanse:8b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q5_K_M":{"mode":"chat","base_model":"aya-expanse:8b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q5_K_S":{"mode":"chat","base_model":"aya-expanse:8b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q6_K":{"mode":"chat","base_model":"aya-expanse:8b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"aya-expanse:8b-q8_0":{"mode":"chat","base_model":"aya-expanse:8b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b":{"mode":"chat","base_model":"granite3-moe:1b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b":{"mode":"chat","base_model":"granite3-moe:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-fp16":{"mode":"chat","base_model":"granite3-moe:1b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q2_K":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q3_K_L":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q3_K_M":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q3_K_S":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q4_0":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q4_1":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q4_K_S":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q5_0":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q5_1":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q5_K_M":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q5_K_S":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q6_K":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:1b-instruct-q8_0":{"mode":"chat","base_model":"granite3-moe:1b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-fp16":{"mode":"chat","base_model":"granite3-moe:3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q2_K":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q3_K_L":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q3_K_M":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q3_K_S":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q4_0":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q4_1":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q4_K_S":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q5_0":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q5_1":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q5_K_M":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q5_K_S":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q6_K":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-moe:3b-instruct-q8_0":{"mode":"chat","base_model":"granite3-moe:3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b":{"mode":"chat","base_model":"nexusraven:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-fp16":{"mode":"chat","base_model":"nexusraven:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q2_K":{"mode":"chat","base_model":"nexusraven:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q3_K_L":{"mode":"chat","base_model":"nexusraven:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q3_K_M":{"mode":"chat","base_model":"nexusraven:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q3_K_S":{"mode":"chat","base_model":"nexusraven:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q4_0":{"mode":"chat","base_model":"nexusraven:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q4_1":{"mode":"chat","base_model":"nexusraven:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q4_K_M":{"mode":"chat","base_model":"nexusraven:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q4_K_S":{"mode":"chat","base_model":"nexusraven:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q5_0":{"mode":"chat","base_model":"nexusraven:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q5_1":{"mode":"chat","base_model":"nexusraven:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q5_K_M":{"mode":"chat","base_model":"nexusraven:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q5_K_S":{"mode":"chat","base_model":"nexusraven:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q6_K":{"mode":"chat","base_model":"nexusraven:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-q8_0":{"mode":"chat","base_model":"nexusraven:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-fp16":{"mode":"chat","base_model":"nexusraven:13b-v2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q2_K":{"mode":"chat","base_model":"nexusraven:13b-v2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q3_K_L":{"mode":"chat","base_model":"nexusraven:13b-v2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q3_K_M":{"mode":"chat","base_model":"nexusraven:13b-v2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q3_K_S":{"mode":"chat","base_model":"nexusraven:13b-v2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q4_0":{"mode":"chat","base_model":"nexusraven:13b-v2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q4_1":{"mode":"chat","base_model":"nexusraven:13b-v2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q4_K_M":{"mode":"chat","base_model":"nexusraven:13b-v2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q4_K_S":{"mode":"chat","base_model":"nexusraven:13b-v2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q5_0":{"mode":"chat","base_model":"nexusraven:13b-v2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q5_1":{"mode":"chat","base_model":"nexusraven:13b-v2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q5_K_M":{"mode":"chat","base_model":"nexusraven:13b-v2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q5_K_S":{"mode":"chat","base_model":"nexusraven:13b-v2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q6_K":{"mode":"chat","base_model":"nexusraven:13b-v2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nexusraven:13b-v2-q8_0":{"mode":"chat","base_model":"nexusraven:13b-v2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:1b":{"mode":"chat","base_model":"falcon3:1b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:3b":{"mode":"chat","base_model":"falcon3:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:7b":{"mode":"chat","base_model":"falcon3:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:10b":{"mode":"chat","base_model":"falcon3:10b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:10b-instruct-fp16":{"mode":"chat","base_model":"falcon3:10b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:10b-instruct-q4_K_M":{"mode":"chat","base_model":"falcon3:10b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:10b-instruct-q8_0":{"mode":"chat","base_model":"falcon3:10b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:1b-instruct-fp16":{"mode":"chat","base_model":"falcon3:1b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:1b-instruct-q4_K_M":{"mode":"chat","base_model":"falcon3:1b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:1b-instruct-q8_0":{"mode":"chat","base_model":"falcon3:1b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:3b-instruct-fp16":{"mode":"chat","base_model":"falcon3:3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:3b-instruct-q4_K_M":{"mode":"chat","base_model":"falcon3:3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:3b-instruct-q8_0":{"mode":"chat","base_model":"falcon3:3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:7b-instruct-fp16":{"mode":"chat","base_model":"falcon3:7b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:7b-instruct-q4_K_M":{"mode":"chat","base_model":"falcon3:7b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon3:7b-instruct-q8_0":{"mode":"chat","base_model":"falcon3:7b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b":{"mode":"chat","base_model":"codeup:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2":{"mode":"chat","base_model":"codeup:13b-llama2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat":{"mode":"chat","base_model":"codeup:13b-llama2-chat","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-fp16":{"mode":"chat","base_model":"codeup:13b-llama2-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q2_K":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q3_K_L":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q3_K_M":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q3_K_S":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q4_0":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q4_1":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q4_K_M":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q4_K_S":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q5_0":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q5_1":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q5_K_M":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q5_K_S":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q6_K":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codeup:13b-llama2-chat-q8_0":{"mode":"chat","base_model":"codeup:13b-llama2-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-fp16":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q2_K":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q3_K_L":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q3_K_M":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q3_K_S":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q4_0":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q4_1":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q4_K_M":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q4_K_S":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q5_0":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q5_1":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q5_K_M":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q5_K_S":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q6_K":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:8x7b-dpo-q8_0":{"mode":"chat","base_model":"nous-hermes2-mixtral:8x7b-dpo-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nous-hermes2-mixtral:dpo":{"mode":"chat","base_model":"nous-hermes2-mixtral:dpo","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b":{"mode":"chat","base_model":"everythinglm:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k":{"mode":"chat","base_model":"everythinglm:13b-16k","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-fp16":{"mode":"chat","base_model":"everythinglm:13b-16k-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q2_K":{"mode":"chat","base_model":"everythinglm:13b-16k-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q3_K_L":{"mode":"chat","base_model":"everythinglm:13b-16k-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q3_K_M":{"mode":"chat","base_model":"everythinglm:13b-16k-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q3_K_S":{"mode":"chat","base_model":"everythinglm:13b-16k-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q4_0":{"mode":"chat","base_model":"everythinglm:13b-16k-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q4_1":{"mode":"chat","base_model":"everythinglm:13b-16k-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q4_K_M":{"mode":"chat","base_model":"everythinglm:13b-16k-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q4_K_S":{"mode":"chat","base_model":"everythinglm:13b-16k-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q5_0":{"mode":"chat","base_model":"everythinglm:13b-16k-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q5_1":{"mode":"chat","base_model":"everythinglm:13b-16k-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q5_K_M":{"mode":"chat","base_model":"everythinglm:13b-16k-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q5_K_S":{"mode":"chat","base_model":"everythinglm:13b-16k-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q6_K":{"mode":"chat","base_model":"everythinglm:13b-16k-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"everythinglm:13b-16k-q8_0":{"mode":"chat","base_model":"everythinglm:13b-16k-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b":{"mode":"chat","base_model":"shieldgemma:2b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b":{"mode":"chat","base_model":"shieldgemma:9b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b":{"mode":"chat","base_model":"shieldgemma:27b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-fp16":{"mode":"chat","base_model":"shieldgemma:27b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q2_K":{"mode":"chat","base_model":"shieldgemma:27b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q3_K_L":{"mode":"chat","base_model":"shieldgemma:27b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q3_K_M":{"mode":"chat","base_model":"shieldgemma:27b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q3_K_S":{"mode":"chat","base_model":"shieldgemma:27b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q4_0":{"mode":"chat","base_model":"shieldgemma:27b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q4_1":{"mode":"chat","base_model":"shieldgemma:27b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q4_K_M":{"mode":"chat","base_model":"shieldgemma:27b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q4_K_S":{"mode":"chat","base_model":"shieldgemma:27b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q5_0":{"mode":"chat","base_model":"shieldgemma:27b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q5_1":{"mode":"chat","base_model":"shieldgemma:27b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q5_K_M":{"mode":"chat","base_model":"shieldgemma:27b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q5_K_S":{"mode":"chat","base_model":"shieldgemma:27b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q6_K":{"mode":"chat","base_model":"shieldgemma:27b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:27b-q8_0":{"mode":"chat","base_model":"shieldgemma:27b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-fp16":{"mode":"chat","base_model":"shieldgemma:2b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q2_K":{"mode":"chat","base_model":"shieldgemma:2b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q3_K_L":{"mode":"chat","base_model":"shieldgemma:2b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q3_K_M":{"mode":"chat","base_model":"shieldgemma:2b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q3_K_S":{"mode":"chat","base_model":"shieldgemma:2b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q4_0":{"mode":"chat","base_model":"shieldgemma:2b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q4_1":{"mode":"chat","base_model":"shieldgemma:2b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q4_K_M":{"mode":"chat","base_model":"shieldgemma:2b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q4_K_S":{"mode":"chat","base_model":"shieldgemma:2b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q5_0":{"mode":"chat","base_model":"shieldgemma:2b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q5_1":{"mode":"chat","base_model":"shieldgemma:2b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q5_K_M":{"mode":"chat","base_model":"shieldgemma:2b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q5_K_S":{"mode":"chat","base_model":"shieldgemma:2b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q6_K":{"mode":"chat","base_model":"shieldgemma:2b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:2b-q8_0":{"mode":"chat","base_model":"shieldgemma:2b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-fp16":{"mode":"chat","base_model":"shieldgemma:9b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q2_K":{"mode":"chat","base_model":"shieldgemma:9b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q3_K_L":{"mode":"chat","base_model":"shieldgemma:9b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q3_K_M":{"mode":"chat","base_model":"shieldgemma:9b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q3_K_S":{"mode":"chat","base_model":"shieldgemma:9b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q4_0":{"mode":"chat","base_model":"shieldgemma:9b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q4_1":{"mode":"chat","base_model":"shieldgemma:9b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q4_K_M":{"mode":"chat","base_model":"shieldgemma:9b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q4_K_S":{"mode":"chat","base_model":"shieldgemma:9b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q5_0":{"mode":"chat","base_model":"shieldgemma:9b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q5_1":{"mode":"chat","base_model":"shieldgemma:9b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q5_K_M":{"mode":"chat","base_model":"shieldgemma:9b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q5_K_S":{"mode":"chat","base_model":"shieldgemma:9b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q6_K":{"mode":"chat","base_model":"shieldgemma:9b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"shieldgemma:9b-q8_0":{"mode":"chat","base_model":"shieldgemma:9b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b":{"mode":"chat","base_model":"granite3.1-moe:1b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b":{"mode":"chat","base_model":"granite3.1-moe:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-fp16":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q2_K":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q3_K_L":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q3_K_M":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q3_K_S":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q4_0":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q4_1":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q4_K_S":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q5_0":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q5_1":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q5_K_M":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q5_K_S":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q6_K":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:1b-instruct-q8_0":{"mode":"chat","base_model":"granite3.1-moe:1b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-fp16":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q2_K":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q3_K_L":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q3_K_M":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q3_K_S":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q4_0":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q4_1":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q4_K_S":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q5_0":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q5_1":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q5_K_M":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q5_K_S":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q6_K":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.1-moe:3b-instruct-q8_0":{"mode":"chat","base_model":"granite3.1-moe:3b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed2:568m":{"mode":"chat","base_model":"snowflake-arctic-embed2:568m","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"snowflake-arctic-embed2:568m-l-fp16":{"mode":"chat","base_model":"snowflake-arctic-embed2:568m-l-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"marco-o1:7b":{"mode":"chat","base_model":"marco-o1:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"marco-o1:7b-fp16":{"mode":"chat","base_model":"marco-o1:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"marco-o1:7b-q4_K_M":{"mode":"chat","base_model":"marco-o1:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"marco-o1:7b-q8_0":{"mode":"chat","base_model":"marco-o1:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b":{"mode":"chat","base_model":"falcon2:11b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-fp16":{"mode":"chat","base_model":"falcon2:11b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q2_K":{"mode":"chat","base_model":"falcon2:11b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q3_K_L":{"mode":"chat","base_model":"falcon2:11b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q3_K_M":{"mode":"chat","base_model":"falcon2:11b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q3_K_S":{"mode":"chat","base_model":"falcon2:11b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q4_0":{"mode":"chat","base_model":"falcon2:11b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q4_1":{"mode":"chat","base_model":"falcon2:11b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q4_K_M":{"mode":"chat","base_model":"falcon2:11b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q4_K_S":{"mode":"chat","base_model":"falcon2:11b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q5_0":{"mode":"chat","base_model":"falcon2:11b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q5_1":{"mode":"chat","base_model":"falcon2:11b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q5_K_M":{"mode":"chat","base_model":"falcon2:11b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q5_K_S":{"mode":"chat","base_model":"falcon2:11b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q6_K":{"mode":"chat","base_model":"falcon2:11b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"falcon2:11b-q8_0":{"mode":"chat","base_model":"falcon2:11b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b":{"mode":"chat","base_model":"mathstral:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-fp16":{"mode":"chat","base_model":"mathstral:7b-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q2_K":{"mode":"chat","base_model":"mathstral:7b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q3_K_L":{"mode":"chat","base_model":"mathstral:7b-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q3_K_M":{"mode":"chat","base_model":"mathstral:7b-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q3_K_S":{"mode":"chat","base_model":"mathstral:7b-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q4_0":{"mode":"chat","base_model":"mathstral:7b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q4_1":{"mode":"chat","base_model":"mathstral:7b-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q4_K_M":{"mode":"chat","base_model":"mathstral:7b-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q4_K_S":{"mode":"chat","base_model":"mathstral:7b-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q5_0":{"mode":"chat","base_model":"mathstral:7b-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q5_1":{"mode":"chat","base_model":"mathstral:7b-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q5_K_M":{"mode":"chat","base_model":"mathstral:7b-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q5_K_S":{"mode":"chat","base_model":"mathstral:7b-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q6_K":{"mode":"chat","base_model":"mathstral:7b-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mathstral:7b-v0.1-q8_0":{"mode":"chat","base_model":"mathstral:7b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b":{"mode":"chat","base_model":"magicoder:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl":{"mode":"chat","base_model":"magicoder:7b-s-cl","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-fp16":{"mode":"chat","base_model":"magicoder:7b-s-cl-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q2_K":{"mode":"chat","base_model":"magicoder:7b-s-cl-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q3_K_L":{"mode":"chat","base_model":"magicoder:7b-s-cl-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q3_K_M":{"mode":"chat","base_model":"magicoder:7b-s-cl-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q3_K_S":{"mode":"chat","base_model":"magicoder:7b-s-cl-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q4_0":{"mode":"chat","base_model":"magicoder:7b-s-cl-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q4_1":{"mode":"chat","base_model":"magicoder:7b-s-cl-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q4_K_M":{"mode":"chat","base_model":"magicoder:7b-s-cl-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q4_K_S":{"mode":"chat","base_model":"magicoder:7b-s-cl-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q5_0":{"mode":"chat","base_model":"magicoder:7b-s-cl-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q5_1":{"mode":"chat","base_model":"magicoder:7b-s-cl-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q5_K_M":{"mode":"chat","base_model":"magicoder:7b-s-cl-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q5_K_S":{"mode":"chat","base_model":"magicoder:7b-s-cl-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q6_K":{"mode":"chat","base_model":"magicoder:7b-s-cl-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"magicoder:7b-s-cl-q8_0":{"mode":"chat","base_model":"magicoder:7b-s-cl-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b":{"mode":"chat","base_model":"stablelm-zephyr:3b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-fp16":{"mode":"chat","base_model":"stablelm-zephyr:3b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q2_K":{"mode":"chat","base_model":"stablelm-zephyr:3b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q3_K_L":{"mode":"chat","base_model":"stablelm-zephyr:3b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q3_K_M":{"mode":"chat","base_model":"stablelm-zephyr:3b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q3_K_S":{"mode":"chat","base_model":"stablelm-zephyr:3b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q4_0":{"mode":"chat","base_model":"stablelm-zephyr:3b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q4_1":{"mode":"chat","base_model":"stablelm-zephyr:3b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q4_K_M":{"mode":"chat","base_model":"stablelm-zephyr:3b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q4_K_S":{"mode":"chat","base_model":"stablelm-zephyr:3b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q5_0":{"mode":"chat","base_model":"stablelm-zephyr:3b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q5_1":{"mode":"chat","base_model":"stablelm-zephyr:3b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q5_K_M":{"mode":"chat","base_model":"stablelm-zephyr:3b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q5_K_S":{"mode":"chat","base_model":"stablelm-zephyr:3b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q6_K":{"mode":"chat","base_model":"stablelm-zephyr:3b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"stablelm-zephyr:3b-q8_0":{"mode":"chat","base_model":"stablelm-zephyr:3b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b":{"mode":"chat","base_model":"reader-lm:0.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b":{"mode":"chat","base_model":"reader-lm:1.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-fp16":{"mode":"chat","base_model":"reader-lm:0.5b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q2_K":{"mode":"chat","base_model":"reader-lm:0.5b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q3_K_L":{"mode":"chat","base_model":"reader-lm:0.5b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q3_K_M":{"mode":"chat","base_model":"reader-lm:0.5b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q3_K_S":{"mode":"chat","base_model":"reader-lm:0.5b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q4_0":{"mode":"chat","base_model":"reader-lm:0.5b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q4_1":{"mode":"chat","base_model":"reader-lm:0.5b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q4_K_M":{"mode":"chat","base_model":"reader-lm:0.5b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q4_K_S":{"mode":"chat","base_model":"reader-lm:0.5b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q5_0":{"mode":"chat","base_model":"reader-lm:0.5b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q5_1":{"mode":"chat","base_model":"reader-lm:0.5b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q5_K_M":{"mode":"chat","base_model":"reader-lm:0.5b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q5_K_S":{"mode":"chat","base_model":"reader-lm:0.5b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q6_K":{"mode":"chat","base_model":"reader-lm:0.5b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:0.5b-q8_0":{"mode":"chat","base_model":"reader-lm:0.5b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-fp16":{"mode":"chat","base_model":"reader-lm:1.5b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q2_K":{"mode":"chat","base_model":"reader-lm:1.5b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q3_K_L":{"mode":"chat","base_model":"reader-lm:1.5b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q3_K_M":{"mode":"chat","base_model":"reader-lm:1.5b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q3_K_S":{"mode":"chat","base_model":"reader-lm:1.5b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q4_0":{"mode":"chat","base_model":"reader-lm:1.5b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q4_1":{"mode":"chat","base_model":"reader-lm:1.5b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q4_K_M":{"mode":"chat","base_model":"reader-lm:1.5b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q4_K_S":{"mode":"chat","base_model":"reader-lm:1.5b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q5_0":{"mode":"chat","base_model":"reader-lm:1.5b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q5_1":{"mode":"chat","base_model":"reader-lm:1.5b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q5_K_M":{"mode":"chat","base_model":"reader-lm:1.5b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q5_K_S":{"mode":"chat","base_model":"reader-lm:1.5b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q6_K":{"mode":"chat","base_model":"reader-lm:1.5b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"reader-lm:1.5b-q8_0":{"mode":"chat","base_model":"reader-lm:1.5b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b":{"mode":"chat","base_model":"solar-pro:22b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-fp16":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q2_K":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q3_K_L":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q3_K_M":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q3_K_S":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q4_0":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q4_1":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q4_K_M":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q4_K_S":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q5_0":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q5_1":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q5_K_M":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q5_K_S":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q6_K":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:22b-preview-instruct-q8_0":{"mode":"chat","base_model":"solar-pro:22b-preview-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"solar-pro:preview":{"mode":"chat","base_model":"solar-pro:preview","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b":{"mode":"chat","base_model":"codebooga:34b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-fp16":{"mode":"chat","base_model":"codebooga:34b-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q2_K":{"mode":"chat","base_model":"codebooga:34b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q3_K_L":{"mode":"chat","base_model":"codebooga:34b-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q3_K_M":{"mode":"chat","base_model":"codebooga:34b-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q3_K_S":{"mode":"chat","base_model":"codebooga:34b-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q4_0":{"mode":"chat","base_model":"codebooga:34b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q4_1":{"mode":"chat","base_model":"codebooga:34b-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q4_K_M":{"mode":"chat","base_model":"codebooga:34b-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q5_0":{"mode":"chat","base_model":"codebooga:34b-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q5_1":{"mode":"chat","base_model":"codebooga:34b-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q5_K_M":{"mode":"chat","base_model":"codebooga:34b-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q5_K_S":{"mode":"chat","base_model":"codebooga:34b-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q6_K":{"mode":"chat","base_model":"codebooga:34b-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"codebooga:34b-v0.1-q8_0":{"mode":"chat","base_model":"codebooga:34b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b":{"mode":"chat","base_model":"duckdb-nsql:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-fp16":{"mode":"chat","base_model":"duckdb-nsql:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q2_K":{"mode":"chat","base_model":"duckdb-nsql:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q3_K_L":{"mode":"chat","base_model":"duckdb-nsql:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q3_K_M":{"mode":"chat","base_model":"duckdb-nsql:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q3_K_S":{"mode":"chat","base_model":"duckdb-nsql:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q4_0":{"mode":"chat","base_model":"duckdb-nsql:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q4_1":{"mode":"chat","base_model":"duckdb-nsql:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q4_K_M":{"mode":"chat","base_model":"duckdb-nsql:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q4_K_S":{"mode":"chat","base_model":"duckdb-nsql:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q5_0":{"mode":"chat","base_model":"duckdb-nsql:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q5_1":{"mode":"chat","base_model":"duckdb-nsql:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q5_K_M":{"mode":"chat","base_model":"duckdb-nsql:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q5_K_S":{"mode":"chat","base_model":"duckdb-nsql:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q6_K":{"mode":"chat","base_model":"duckdb-nsql:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"duckdb-nsql:7b-q8_0":{"mode":"chat","base_model":"duckdb-nsql:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b":{"mode":"chat","base_model":"mistrallite:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-fp16":{"mode":"chat","base_model":"mistrallite:7b-v0.1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q2_K":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q3_K_L":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q3_K_M":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q3_K_S":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q4_0":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q4_1":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q4_K_M":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q4_K_S":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q5_0":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q5_1":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q5_K_M":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q5_K_S":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q6_K":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistrallite:7b-v0.1-q8_0":{"mode":"chat","base_model":"mistrallite:7b-v0.1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-fp16":{"mode":"chat","base_model":"llama-guard3:1b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q2_K":{"mode":"chat","base_model":"llama-guard3:1b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q3_K_L":{"mode":"chat","base_model":"llama-guard3:1b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q3_K_M":{"mode":"chat","base_model":"llama-guard3:1b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q3_K_S":{"mode":"chat","base_model":"llama-guard3:1b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q4_0":{"mode":"chat","base_model":"llama-guard3:1b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q4_1":{"mode":"chat","base_model":"llama-guard3:1b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q4_K_M":{"mode":"chat","base_model":"llama-guard3:1b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q4_K_S":{"mode":"chat","base_model":"llama-guard3:1b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q5_0":{"mode":"chat","base_model":"llama-guard3:1b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q5_1":{"mode":"chat","base_model":"llama-guard3:1b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q5_K_M":{"mode":"chat","base_model":"llama-guard3:1b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q5_K_S":{"mode":"chat","base_model":"llama-guard3:1b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q6_K":{"mode":"chat","base_model":"llama-guard3:1b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:1b-q8_0":{"mode":"chat","base_model":"llama-guard3:1b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-fp16":{"mode":"chat","base_model":"llama-guard3:8b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q2_K":{"mode":"chat","base_model":"llama-guard3:8b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q3_K_L":{"mode":"chat","base_model":"llama-guard3:8b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q3_K_M":{"mode":"chat","base_model":"llama-guard3:8b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q3_K_S":{"mode":"chat","base_model":"llama-guard3:8b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q4_0":{"mode":"chat","base_model":"llama-guard3:8b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q4_1":{"mode":"chat","base_model":"llama-guard3:8b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q4_K_M":{"mode":"chat","base_model":"llama-guard3:8b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q4_K_S":{"mode":"chat","base_model":"llama-guard3:8b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q5_0":{"mode":"chat","base_model":"llama-guard3:8b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q5_1":{"mode":"chat","base_model":"llama-guard3:8b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q5_K_M":{"mode":"chat","base_model":"llama-guard3:8b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q5_K_S":{"mode":"chat","base_model":"llama-guard3:8b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q6_K":{"mode":"chat","base_model":"llama-guard3:8b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"llama-guard3:8b-q8_0":{"mode":"chat","base_model":"llama-guard3:8b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b":{"mode":"chat","base_model":"wizard-vicuna:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-fp16":{"mode":"chat","base_model":"wizard-vicuna:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q2_K":{"mode":"chat","base_model":"wizard-vicuna:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q3_K_L":{"mode":"chat","base_model":"wizard-vicuna:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q3_K_M":{"mode":"chat","base_model":"wizard-vicuna:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q3_K_S":{"mode":"chat","base_model":"wizard-vicuna:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q4_0":{"mode":"chat","base_model":"wizard-vicuna:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q4_1":{"mode":"chat","base_model":"wizard-vicuna:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q4_K_M":{"mode":"chat","base_model":"wizard-vicuna:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q4_K_S":{"mode":"chat","base_model":"wizard-vicuna:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q5_0":{"mode":"chat","base_model":"wizard-vicuna:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q5_1":{"mode":"chat","base_model":"wizard-vicuna:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q5_K_M":{"mode":"chat","base_model":"wizard-vicuna:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q5_K_S":{"mode":"chat","base_model":"wizard-vicuna:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q6_K":{"mode":"chat","base_model":"wizard-vicuna:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"wizard-vicuna:13b-q8_0":{"mode":"chat","base_model":"wizard-vicuna:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:2.4b":{"mode":"chat","base_model":"exaone3.5:2.4b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:7.8b":{"mode":"chat","base_model":"exaone3.5:7.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:32b":{"mode":"chat","base_model":"exaone3.5:32b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:2.4b-instruct-fp16":{"mode":"chat","base_model":"exaone3.5:2.4b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:2.4b-instruct-q4_K_M":{"mode":"chat","base_model":"exaone3.5:2.4b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:2.4b-instruct-q8_0":{"mode":"chat","base_model":"exaone3.5:2.4b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:32b-instruct-fp16":{"mode":"chat","base_model":"exaone3.5:32b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:32b-instruct-q4_K_M":{"mode":"chat","base_model":"exaone3.5:32b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:32b-instruct-q8_0":{"mode":"chat","base_model":"exaone3.5:32b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:7.8b-instruct-fp16":{"mode":"chat","base_model":"exaone3.5:7.8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:7.8b-instruct-q4_K_M":{"mode":"chat","base_model":"exaone3.5:7.8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"exaone3.5:7.8b-instruct-q8_0":{"mode":"chat","base_model":"exaone3.5:7.8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b":{"mode":"chat","base_model":"megadolphin:120b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2":{"mode":"chat","base_model":"megadolphin:120b-v2.2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-fp16":{"mode":"chat","base_model":"megadolphin:120b-v2.2-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q2_K":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q3_K_L":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q3_K_M":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q3_K_S":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q4_0":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q4_1":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q4_K_M":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q4_K_S":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q5_0":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q5_1":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q5_K_M":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q5_K_S":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q6_K":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:120b-v2.2-q8_0":{"mode":"chat","base_model":"megadolphin:120b-v2.2-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"megadolphin:v2.2":{"mode":"chat","base_model":"megadolphin:v2.2","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b":{"mode":"chat","base_model":"nuextract:3.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-fp16":{"mode":"chat","base_model":"nuextract:3.8b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q2_K":{"mode":"chat","base_model":"nuextract:3.8b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q3_K_L":{"mode":"chat","base_model":"nuextract:3.8b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q3_K_M":{"mode":"chat","base_model":"nuextract:3.8b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q3_K_S":{"mode":"chat","base_model":"nuextract:3.8b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q4_0":{"mode":"chat","base_model":"nuextract:3.8b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q4_1":{"mode":"chat","base_model":"nuextract:3.8b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q4_K_M":{"mode":"chat","base_model":"nuextract:3.8b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q4_K_S":{"mode":"chat","base_model":"nuextract:3.8b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q5_0":{"mode":"chat","base_model":"nuextract:3.8b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q5_1":{"mode":"chat","base_model":"nuextract:3.8b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q5_K_M":{"mode":"chat","base_model":"nuextract:3.8b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q5_K_S":{"mode":"chat","base_model":"nuextract:3.8b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q6_K":{"mode":"chat","base_model":"nuextract:3.8b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"nuextract:3.8b-q8_0":{"mode":"chat","base_model":"nuextract:3.8b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"opencoder:1.5b":{"mode":"chat","base_model":"opencoder:1.5b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"opencoder:8b":{"mode":"chat","base_model":"opencoder:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"opencoder:1.5b-instruct-fp16":{"mode":"chat","base_model":"opencoder:1.5b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"opencoder:1.5b-instruct-q4_K_M":{"mode":"chat","base_model":"opencoder:1.5b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"opencoder:1.5b-instruct-q8_0":{"mode":"chat","base_model":"opencoder:1.5b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"opencoder:8b-instruct-fp16":{"mode":"chat","base_model":"opencoder:8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"opencoder:8b-instruct-q4_K_M":{"mode":"chat","base_model":"opencoder:8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"opencoder:8b-instruct-q8_0":{"mode":"chat","base_model":"opencoder:8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b":{"mode":"chat","base_model":"notux:8x7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1":{"mode":"chat","base_model":"notux:8x7b-v1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-fp16":{"mode":"chat","base_model":"notux:8x7b-v1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q2_K":{"mode":"chat","base_model":"notux:8x7b-v1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q3_K_L":{"mode":"chat","base_model":"notux:8x7b-v1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q3_K_M":{"mode":"chat","base_model":"notux:8x7b-v1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q3_K_S":{"mode":"chat","base_model":"notux:8x7b-v1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q4_0":{"mode":"chat","base_model":"notux:8x7b-v1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q4_1":{"mode":"chat","base_model":"notux:8x7b-v1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q4_K_M":{"mode":"chat","base_model":"notux:8x7b-v1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q4_K_S":{"mode":"chat","base_model":"notux:8x7b-v1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q5_0":{"mode":"chat","base_model":"notux:8x7b-v1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q5_1":{"mode":"chat","base_model":"notux:8x7b-v1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q5_K_M":{"mode":"chat","base_model":"notux:8x7b-v1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q5_K_S":{"mode":"chat","base_model":"notux:8x7b-v1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q6_K":{"mode":"chat","base_model":"notux:8x7b-v1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notux:8x7b-v1-q8_0":{"mode":"chat","base_model":"notux:8x7b-v1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b":{"mode":"chat","base_model":"open-orca-platypus2:13b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-fp16":{"mode":"chat","base_model":"open-orca-platypus2:13b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q2_K":{"mode":"chat","base_model":"open-orca-platypus2:13b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q3_K_L":{"mode":"chat","base_model":"open-orca-platypus2:13b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q3_K_M":{"mode":"chat","base_model":"open-orca-platypus2:13b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q3_K_S":{"mode":"chat","base_model":"open-orca-platypus2:13b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q4_0":{"mode":"chat","base_model":"open-orca-platypus2:13b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q4_1":{"mode":"chat","base_model":"open-orca-platypus2:13b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q4_K_M":{"mode":"chat","base_model":"open-orca-platypus2:13b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q4_K_S":{"mode":"chat","base_model":"open-orca-platypus2:13b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q5_0":{"mode":"chat","base_model":"open-orca-platypus2:13b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q5_1":{"mode":"chat","base_model":"open-orca-platypus2:13b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q5_K_M":{"mode":"chat","base_model":"open-orca-platypus2:13b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q5_K_S":{"mode":"chat","base_model":"open-orca-platypus2:13b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q6_K":{"mode":"chat","base_model":"open-orca-platypus2:13b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"open-orca-platypus2:13b-q8_0":{"mode":"chat","base_model":"open-orca-platypus2:13b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b":{"mode":"chat","base_model":"notus:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1":{"mode":"chat","base_model":"notus:7b-v1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-fp16":{"mode":"chat","base_model":"notus:7b-v1-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q2_K":{"mode":"chat","base_model":"notus:7b-v1-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q3_K_L":{"mode":"chat","base_model":"notus:7b-v1-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q3_K_M":{"mode":"chat","base_model":"notus:7b-v1-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q3_K_S":{"mode":"chat","base_model":"notus:7b-v1-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q4_0":{"mode":"chat","base_model":"notus:7b-v1-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q4_1":{"mode":"chat","base_model":"notus:7b-v1-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q4_K_M":{"mode":"chat","base_model":"notus:7b-v1-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q4_K_S":{"mode":"chat","base_model":"notus:7b-v1-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q5_0":{"mode":"chat","base_model":"notus:7b-v1-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q5_1":{"mode":"chat","base_model":"notus:7b-v1-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q5_K_M":{"mode":"chat","base_model":"notus:7b-v1-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q5_K_S":{"mode":"chat","base_model":"notus:7b-v1-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q6_K":{"mode":"chat","base_model":"notus:7b-v1-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"notus:7b-v1-q8_0":{"mode":"chat","base_model":"notus:7b-v1-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-fp16":{"mode":"chat","base_model":"goliath:120b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q2_K":{"mode":"chat","base_model":"goliath:120b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q3_K_L":{"mode":"chat","base_model":"goliath:120b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q3_K_M":{"mode":"chat","base_model":"goliath:120b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q3_K_S":{"mode":"chat","base_model":"goliath:120b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q4_0":{"mode":"chat","base_model":"goliath:120b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q4_1":{"mode":"chat","base_model":"goliath:120b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q4_K_M":{"mode":"chat","base_model":"goliath:120b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q4_K_S":{"mode":"chat","base_model":"goliath:120b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q5_0":{"mode":"chat","base_model":"goliath:120b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q5_1":{"mode":"chat","base_model":"goliath:120b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q5_K_M":{"mode":"chat","base_model":"goliath:120b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q5_K_S":{"mode":"chat","base_model":"goliath:120b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q6_K":{"mode":"chat","base_model":"goliath:120b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"goliath:120b-q8_0":{"mode":"chat","base_model":"goliath:120b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b":{"mode":"chat","base_model":"bespoke-minicheck:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-fp16":{"mode":"chat","base_model":"bespoke-minicheck:7b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q2_K":{"mode":"chat","base_model":"bespoke-minicheck:7b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q3_K_L":{"mode":"chat","base_model":"bespoke-minicheck:7b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q3_K_M":{"mode":"chat","base_model":"bespoke-minicheck:7b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q3_K_S":{"mode":"chat","base_model":"bespoke-minicheck:7b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q4_0":{"mode":"chat","base_model":"bespoke-minicheck:7b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q4_1":{"mode":"chat","base_model":"bespoke-minicheck:7b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q4_K_M":{"mode":"chat","base_model":"bespoke-minicheck:7b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q4_K_S":{"mode":"chat","base_model":"bespoke-minicheck:7b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q5_0":{"mode":"chat","base_model":"bespoke-minicheck:7b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q5_1":{"mode":"chat","base_model":"bespoke-minicheck:7b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q5_K_M":{"mode":"chat","base_model":"bespoke-minicheck:7b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q5_K_S":{"mode":"chat","base_model":"bespoke-minicheck:7b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q6_K":{"mode":"chat","base_model":"bespoke-minicheck:7b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bespoke-minicheck:7b-q8_0":{"mode":"chat","base_model":"bespoke-minicheck:7b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r7b:7b":{"mode":"chat","base_model":"command-r7b:7b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r7b:7b-12-2024-fp16":{"mode":"chat","base_model":"command-r7b:7b-12-2024-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r7b:7b-12-2024-q4_K_M":{"mode":"chat","base_model":"command-r7b:7b-12-2024-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"command-r7b:7b-12-2024-q8_0":{"mode":"chat","base_model":"command-r7b:7b-12-2024-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b":{"mode":"chat","base_model":"firefunction-v2:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-fp16":{"mode":"chat","base_model":"firefunction-v2:70b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q2_K":{"mode":"chat","base_model":"firefunction-v2:70b-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q3_K_L":{"mode":"chat","base_model":"firefunction-v2:70b-q3_K_L","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q3_K_M":{"mode":"chat","base_model":"firefunction-v2:70b-q3_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q3_K_S":{"mode":"chat","base_model":"firefunction-v2:70b-q3_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q4_0":{"mode":"chat","base_model":"firefunction-v2:70b-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q4_1":{"mode":"chat","base_model":"firefunction-v2:70b-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q4_K_M":{"mode":"chat","base_model":"firefunction-v2:70b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q4_K_S":{"mode":"chat","base_model":"firefunction-v2:70b-q4_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q5_0":{"mode":"chat","base_model":"firefunction-v2:70b-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q5_1":{"mode":"chat","base_model":"firefunction-v2:70b-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q5_K_M":{"mode":"chat","base_model":"firefunction-v2:70b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q5_K_S":{"mode":"chat","base_model":"firefunction-v2:70b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q6_K":{"mode":"chat","base_model":"firefunction-v2:70b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"firefunction-v2:70b-q8_0":{"mode":"chat","base_model":"firefunction-v2:70b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tulu3:8b":{"mode":"chat","base_model":"tulu3:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tulu3:70b":{"mode":"chat","base_model":"tulu3:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tulu3:70b-fp16":{"mode":"chat","base_model":"tulu3:70b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tulu3:70b-q4_K_M":{"mode":"chat","base_model":"tulu3:70b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tulu3:70b-q8_0":{"mode":"chat","base_model":"tulu3:70b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tulu3:8b-fp16":{"mode":"chat","base_model":"tulu3:8b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tulu3:8b-q4_K_M":{"mode":"chat","base_model":"tulu3:8b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"tulu3:8b-q8_0":{"mode":"chat","base_model":"tulu3:8b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dbrx:132b":{"mode":"chat","base_model":"dbrx:132b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dbrx:132b-instruct-fp16":{"mode":"chat","base_model":"dbrx:132b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dbrx:132b-instruct-q2_K":{"mode":"chat","base_model":"dbrx:132b-instruct-q2_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dbrx:132b-instruct-q4_0":{"mode":"chat","base_model":"dbrx:132b-instruct-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"dbrx:132b-instruct-q8_0":{"mode":"chat","base_model":"dbrx:132b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-embedding:30m":{"mode":"embedding","base_model":"granite-embedding:30m","provider":"ollama","max_input_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-embedding:278m":{"mode":"embedding","base_model":"granite-embedding:278m","provider":"ollama","max_input_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-embedding:278m-fp16":{"mode":"embedding","base_model":"granite-embedding:278m-fp16","provider":"ollama","max_input_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-embedding:30m-en":{"mode":"embedding","base_model":"granite-embedding:30m-en","provider":"ollama","max_input_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite-embedding:30m-en-fp16":{"mode":"embedding","base_model":"granite-embedding:30m-en-fp16","provider":"ollama","max_input_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:2b":{"mode":"chat","base_model":"granite3-guardian:2b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:8b":{"mode":"chat","base_model":"granite3-guardian:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:2b-fp16":{"mode":"chat","base_model":"granite3-guardian:2b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:2b-q8_0":{"mode":"chat","base_model":"granite3-guardian:2b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:8b-fp16":{"mode":"chat","base_model":"granite3-guardian:8b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:8b-q5_K_M":{"mode":"chat","base_model":"granite3-guardian:8b-q5_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:8b-q5_K_S":{"mode":"chat","base_model":"granite3-guardian:8b-q5_K_S","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:8b-q6_K":{"mode":"chat","base_model":"granite3-guardian:8b-q6_K","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3-guardian:8b-q8_0":{"mode":"chat","base_model":"granite3-guardian:8b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"alfred:40b":{"mode":"chat","base_model":"alfred:40b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"alfred:40b-1023-q4_0":{"mode":"chat","base_model":"alfred:40b-1023-q4_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"alfred:40b-1023-q4_1":{"mode":"chat","base_model":"alfred:40b-1023-q4_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"alfred:40b-1023-q5_0":{"mode":"chat","base_model":"alfred:40b-1023-q5_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"alfred:40b-1023-q5_1":{"mode":"chat","base_model":"alfred:40b-1023-q5_1","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"alfred:40b-1023-q8_0":{"mode":"chat","base_model":"alfred:40b-1023-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:1b":{"mode":"chat","base_model":"sailor2:1b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:8b":{"mode":"chat","base_model":"sailor2:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:20b":{"mode":"chat","base_model":"sailor2:20b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:1b-chat-fp16":{"mode":"chat","base_model":"sailor2:1b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:1b-chat-q4_K_M":{"mode":"chat","base_model":"sailor2:1b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:1b-chat-q8_0":{"mode":"chat","base_model":"sailor2:1b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:20b-chat-fp16":{"mode":"chat","base_model":"sailor2:20b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:20b-chat-q4_K_M":{"mode":"chat","base_model":"sailor2:20b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:20b-chat-q8_0":{"mode":"chat","base_model":"sailor2:20b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:8b-chat-fp16":{"mode":"chat","base_model":"sailor2:8b-chat-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:8b-chat-q4_K_M":{"mode":"chat","base_model":"sailor2:8b-chat-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"sailor2:8b-chat-q8_0":{"mode":"chat","base_model":"sailor2:8b-chat-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"r1-1776:70b":{"mode":"chat","base_model":"r1-1776:70b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"r1-1776:671b":{"mode":"chat","base_model":"r1-1776:671b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"r1-1776:671b-fp16":{"mode":"chat","base_model":"r1-1776:671b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"r1-1776:671b-q4_K_M":{"mode":"chat","base_model":"r1-1776:671b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"r1-1776:671b-q8_0":{"mode":"chat","base_model":"r1-1776:671b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"r1-1776:70b-distill-llama-fp16":{"mode":"chat","base_model":"r1-1776:70b-distill-llama-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"r1-1776:70b-distill-llama-q4_K_M":{"mode":"chat","base_model":"r1-1776:70b-distill-llama-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"r1-1776:70b-distill-llama-q8_0":{"mode":"chat","base_model":"r1-1776:70b-distill-llama-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2:2b":{"mode":"chat","base_model":"granite3.2:2b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2:8b":{"mode":"chat","base_model":"granite3.2:8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2:2b-instruct-fp16":{"mode":"chat","base_model":"granite3.2:2b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2:2b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3.2:2b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2:2b-instruct-q8_0":{"mode":"chat","base_model":"granite3.2:2b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2:8b-instruct-fp16":{"mode":"chat","base_model":"granite3.2:8b-instruct-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2:8b-instruct-q4_K_M":{"mode":"chat","base_model":"granite3.2:8b-instruct-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2:8b-instruct-q8_0":{"mode":"chat","base_model":"granite3.2:8b-instruct-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"granite3.2-vision:2b":{"mode":"chat","base_model":"granite3.2-vision:2b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"granite3.2-vision:2b-fp16":{"mode":"chat","base_model":"granite3.2-vision:2b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"granite3.2-vision:2b-q4_K_M":{"mode":"chat","base_model":"granite3.2-vision:2b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"granite3.2-vision:2b-q8_0":{"mode":"chat","base_model":"granite3.2-vision:2b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"phi4-mini:3.8b":{"mode":"chat","base_model":"phi4-mini:3.8b","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi4-mini:3.8b-fp16":{"mode":"chat","base_model":"phi4-mini:3.8b-fp16","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi4-mini:3.8b-q4_K_M":{"mode":"chat","base_model":"phi4-mini:3.8b-q4_K_M","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"phi4-mini:3.8b-q8_0":{"mode":"chat","base_model":"phi4-mini:3.8b-q8_0","provider":"ollama","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":64000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-3.5-turbo-0125":{"mode":"chat","base_model":"openai/gpt-3.5-turbo-0125","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-3.5-turbo-1106":{"mode":"chat","base_model":"openai/gpt-3.5-turbo-1106","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"openai/gpt-3.5-turbo":{"mode":"chat","base_model":"openai/gpt-3.5-turbo","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4":{"mode":"chat","base_model":"openai/gpt-4","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4-0613":{"mode":"chat","base_model":"openai/gpt-4-0613","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4-0125-preview":{"mode":"chat","base_model":"openai/gpt-4-0125-preview","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4-vision-preview":{"mode":"image_generation","base_model":"openai/gpt-4-vision-preview","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4-1106-vision-preview":{"mode":"image_generation","base_model":"openai/gpt-4-1106-vision-preview","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o":{"mode":"image_generation","base_model":"openai/gpt-4o","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv","supports_web_search":true},"openai/gpt-4o-2024-05-13":{"mode":"image_generation","base_model":"openai/gpt-4o-2024-05-13","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"openai/o1-preview":{"mode":"chat","base_model":"openai/o1-preview","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o1":{"mode":"image_generation","base_model":"openai/o1","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o1-preview-2024-09-12":{"mode":"chat","base_model":"openai/o1-preview-2024-09-12","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o3-mini":{"mode":"chat","base_model":"openai/o3-mini","provider":"litellm","max_input_tokens":100000,"max_output_tokens":100000,"max_tokens":100000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4.5-preview":{"mode":"image_generation","base_model":"openai/gpt-4.5-preview","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4.5-preview-2025-02-27":{"mode":"image_generation","base_model":"openai/gpt-4.5-preview-2025-02-27","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4.1-2025-04-14":{"mode":"image_generation","base_model":"openai/gpt-4.1-2025-04-14","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4.1":{"mode":"image_generation","base_model":"openai/gpt-4.1","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4.1-mini-2025-04-14":{"mode":"image_generation","base_model":"openai/gpt-4.1-mini-2025-04-14","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4.1-mini":{"mode":"image_generation","base_model":"openai/gpt-4.1-mini","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4.1-nano-2025-04-14":{"mode":"image_generation","base_model":"openai/gpt-4.1-nano-2025-04-14","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4.1-nano":{"mode":"image_generation","base_model":"openai/gpt-4.1-nano","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":32768}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"openai/o3-2025-04-16":{"mode":"image_generation","base_model":"openai/o3-2025-04-16","provider":"litellm","max_input_tokens":100000,"max_output_tokens":100000,"max_tokens":100000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o3-mini-2025-01-31":{"mode":"image_generation","base_model":"openai/o3-mini-2025-01-31","provider":"litellm","max_input_tokens":100000,"max_output_tokens":100000,"max_tokens":100000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o4-mini-2025-04-16":{"mode":"chat","base_model":"openai/o4-mini-2025-04-16","provider":"litellm","max_input_tokens":100000,"max_output_tokens":100000,"max_tokens":100000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","default":"medium","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o1-mini-2024-09-12":{"mode":"chat","base_model":"openai/o1-mini-2024-09-12","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":65536}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o1-mini":{"mode":"chat","base_model":"openai/o1-mini","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":65536}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/o1-2024-12-17":{"mode":"image_generation","base_model":"openai/o1-2024-12-17","provider":"litellm","max_input_tokens":100000,"max_output_tokens":100000,"max_tokens":100000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":100000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":100000,"range":{"min":1,"max":100000}}],"source":"merged_from_llm_models_csv"},"openai/gpt-4o-audio-preview":{"mode":"chat","base_model":"openai/gpt-4o-audio-preview","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o-audio-preview-2025-06-03":{"mode":"chat","base_model":"openai/gpt-4o-audio-preview-2025-06-03","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o-audio-preview-2024-12-17":{"mode":"chat","base_model":"openai/gpt-4o-audio-preview-2024-12-17","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o-audio-preview-2024-10-01":{"mode":"chat","base_model":"openai/gpt-4o-audio-preview-2024-10-01","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o-mini-audio-preview":{"mode":"chat","base_model":"openai/gpt-4o-mini-audio-preview","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o-realtime-preview-2025-06-03":{"mode":"chat","base_model":"openai/gpt-4o-realtime-preview-2025-06-03","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"instructions","label":"Instructions","helpText":"The default system instructions (i.e. system message) prepended to model calls.","type":"text"},{"id":"voice","label":"Voice","helpText":"Pre-selected voice used when generating the audio","type":"select","default":"alloy","options":[{"label":"Alloy","value":"alloy"},{"label":"Ash","value":"ash"},{"label":"Ballad","value":"ballad"},{"label":"Coral","value":"coral"},{"label":"Echo","value":"echo"},{"label":"Sage","value":"sage"},{"label":"Shimmer","value":"shimmer"},{"label":"Verse","value":"verse"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0.6,"max":1.2,"step":0.01}},{"id":"max_response_output_tokens","label":"Max Response Output Tokens","helpText":"Maximum number of output tokens for a single assistant response, inclusive of tool calls.","type":"number","default":510},{"id":"input_audio_noise_reduction","label":"Input Audio Noise Reduction","helpText":"Noise reduction applied to audio input, helpful with VAD and model understanding.","type":"select","accesorKey":"type","options":[{"label":"None","value":"none"},{"label":"Near Field","value":"near_field"},{"label":"Far Field","value":"far_field"}]},{"id":"speed","label":"Speed","helpText":"The speed of the model's spoken response. ","type":"number","default":1,"range":{"min":0.25,"max":1.5,"step":0.05}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-4o-realtime-preview":{"mode":"chat","base_model":"openai/gpt-4o-realtime-preview","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"instructions","label":"Instructions","helpText":"The default system instructions (i.e. system message) prepended to model calls.","type":"text"},{"id":"voice","label":"Voice","helpText":"Pre-selected voice used when generating the audio","type":"select","default":"alloy","options":[{"label":"Alloy","value":"alloy"},{"label":"Ash","value":"ash"},{"label":"Ballad","value":"ballad"},{"label":"Coral","value":"coral"},{"label":"Echo","value":"echo"},{"label":"Sage","value":"sage"},{"label":"Shimmer","value":"shimmer"},{"label":"Verse","value":"verse"}]},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0.6,"max":1.2,"step":0.01}},{"id":"max_response_output_tokens","label":"Max Response Output Tokens","helpText":"Maximum number of output tokens for a single assistant response, inclusive of tool calls.","type":"number","default":510},{"id":"input_audio_noise_reduction","label":"Input Audio Noise Reduction","helpText":"Noise reduction applied to audio input, helpful with VAD and model understanding.","type":"select","accesorKey":"type","options":[{"label":"None","value":"none"},{"label":"Near Field","value":"near_field"},{"label":"Far Field","value":"far_field"}]},{"id":"speed","label":"Speed","helpText":"The speed of the model's spoken response. ","type":"number","default":1,"range":{"min":0.25,"max":1.5,"step":0.05}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openai/gpt-5-2025-08-07":{"mode":"image_generation","base_model":"openai/gpt-5-2025-08-07","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-5":{"mode":"image_generation","base_model":"openai/gpt-5","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-5-mini":{"mode":"image_generation","base_model":"openai/gpt-5-mini","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-5-mini-2025-08-07":{"mode":"image_generation","base_model":"openai/gpt-5-mini-2025-08-07","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-5-nano":{"mode":"image_generation","base_model":"openai/gpt-5-nano","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-5-nano-2025-08-07":{"mode":"image_generation","base_model":"openai/gpt-5-nano-2025-08-07","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-5-chat":{"mode":"image_generation","base_model":"openai/gpt-5-chat","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-5-chat-latest":{"mode":"image_generation","base_model":"openai/gpt-5-chat-latest","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openai/gpt-oss-120b":{"mode":"chat","base_model":"openai/gpt-oss-120b","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_output_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":131072}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"together/openai/gpt-oss-120b":{"mode":"chat","base_model":"together/openai/gpt-oss-120b","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"together/zero-one-ai/Yi-34B-Chat":{"mode":"chat","base_model":"together/zero-one-ai/Yi-34B-Chat","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Presence Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Reasoning effort is used to give the model guidance on how many reasoning tokens it should generate before creating a response to the prompt. You can specify one of low, medium, or high for this parameter, where low will favor speed and economical token usage, and high will favor more complete reasoning at the cost of more tokens generated and slower responses. The default value is medium, which is a balance between speed and reasoning accuracy.","type":"select","accesorKey":"type","default":{"type":"medium"},"options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"source":"merged_from_llm_models_csv"},"together/Austism/chronos-hermes-13b":{"mode":"chat","base_model":"together/Austism/chronos-hermes-13b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/DiscoResearch/DiscoLM-mixtral-8x7b-v2":{"mode":"chat","base_model":"together/DiscoResearch/DiscoLM-mixtral-8x7b-v2","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Gryphe/MythoMax-L2-13b":{"mode":"chat","base_model":"together/Gryphe/MythoMax-L2-13b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/lmsys/vicuna-13b-v1.5":{"mode":"chat","base_model":"together/lmsys/vicuna-13b-v1.5","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/lmsys/vicuna-7b-v1.5":{"mode":"chat","base_model":"together/lmsys/vicuna-7b-v1.5","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/lmsys/vicuna-13b-v1.5-16k":{"mode":"chat","base_model":"together/lmsys/vicuna-13b-v1.5-16k","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-13b-Instruct-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-13b-Instruct-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-34b-Instruct-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-34b-Instruct-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-70b-Instruct-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-70b-Instruct-hf","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-7b-Instruct-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-7b-Instruct-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/llama-2-13b-chat":{"mode":"chat","base_model":"together/togethercomputer/llama-2-13b-chat","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/llama-2-70b-chat":{"mode":"chat","base_model":"together/togethercomputer/llama-2-70b-chat","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/llama-2-7b-chat":{"mode":"chat","base_model":"together/togethercomputer/llama-2-7b-chat","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Capybara-7B-V1p9":{"mode":"chat","base_model":"together/NousResearch/Nous-Capybara-7B-V1p9","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Hermes-2-Mixtral-8x7B-DPO":{"mode":"chat","base_model":"together/NousResearch/Nous-Hermes-2-Mixtral-8x7B-DPO","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Hermes-2-Mixtral-8x7B-SFT":{"mode":"chat","base_model":"together/NousResearch/Nous-Hermes-2-Mixtral-8x7B-SFT","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Hermes-Llama2-70b":{"mode":"chat","base_model":"together/NousResearch/Nous-Hermes-Llama2-70b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Hermes-llama-2-7b":{"mode":"chat","base_model":"together/NousResearch/Nous-Hermes-llama-2-7b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Hermes-Llama2-13b":{"mode":"chat","base_model":"together/NousResearch/Nous-Hermes-Llama2-13b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Hermes-2-Yi-34B":{"mode":"chat","base_model":"together/NousResearch/Nous-Hermes-2-Yi-34B","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/openchat/openchat-3.5-1210":{"mode":"chat","base_model":"together/openchat/openchat-3.5-1210","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Open-Orca/Mistral-7B-OpenOrca":{"mode":"chat","base_model":"together/Open-Orca/Mistral-7B-OpenOrca","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/Qwen-7B-Chat":{"mode":"chat","base_model":"together/togethercomputer/Qwen-7B-Chat","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/snorkelai/Snorkel-Mistral-PairRM-DPO":{"mode":"chat","base_model":"together/snorkelai/Snorkel-Mistral-PairRM-DPO","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/alpaca-7b":{"mode":"chat","base_model":"together/togethercomputer/alpaca-7b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/falcon-40b-instruct":{"mode":"chat","base_model":"together/togethercomputer/falcon-40b-instruct","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/falcon-7b-instruct":{"mode":"chat","base_model":"together/togethercomputer/falcon-7b-instruct","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/GPT-NeoXT-Chat-Base-20B":{"mode":"chat","base_model":"together/togethercomputer/GPT-NeoXT-Chat-Base-20B","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/Llama-2-7B-32K-Instruct":{"mode":"chat","base_model":"together/togethercomputer/Llama-2-7B-32K-Instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/Pythia-Chat-Base-7B-v0.16":{"mode":"chat","base_model":"together/togethercomputer/Pythia-Chat-Base-7B-v0.16","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/RedPajama-INCITE-Chat-3B-v1":{"mode":"chat","base_model":"together/togethercomputer/RedPajama-INCITE-Chat-3B-v1","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/RedPajama-INCITE-7B-Chat":{"mode":"chat","base_model":"together/togethercomputer/RedPajama-INCITE-7B-Chat","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/StripedHyena-Nous-7B":{"mode":"chat","base_model":"together/togethercomputer/StripedHyena-Nous-7B","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Undi95/ReMM-SLERP-L2-13B":{"mode":"chat","base_model":"together/Undi95/ReMM-SLERP-L2-13B","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Undi95/Toppy-M-7B":{"mode":"chat","base_model":"together/Undi95/Toppy-M-7B","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/WizardLM/WizardLM-13B-V1.2":{"mode":"chat","base_model":"together/WizardLM/WizardLM-13B-V1.2","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/garage-bAInd/Platypus2-70B-instruct":{"mode":"chat","base_model":"together/garage-bAInd/Platypus2-70B-instruct","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/mistralai/Mistral-7B-Instruct-v0.1":{"mode":"chat","base_model":"together/mistralai/Mistral-7B-Instruct-v0.1","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","type":"select","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call."},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"together/mistralai/Mistral-7B-Instruct-v0.2":{"mode":"chat","base_model":"together/mistralai/Mistral-7B-Instruct-v0.2","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/mistralai/Mixtral-8x7B-Instruct-v0.1":{"mode":"chat","base_model":"together/mistralai/Mixtral-8x7B-Instruct-v0.1","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","type":"select","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call."},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"together/teknium/OpenHermes-2-Mistral-7B":{"mode":"chat","base_model":"together/teknium/OpenHermes-2-Mistral-7B","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/teknium/OpenHermes-2p5-Mistral-7B":{"mode":"chat","base_model":"together/teknium/OpenHermes-2p5-Mistral-7B","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/upstage/SOLAR-10.7B-Instruct-v1.0":{"mode":"chat","base_model":"together/upstage/SOLAR-10.7B-Instruct-v1.0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/zero-one-ai/Yi-34B":{"mode":"chat","base_model":"together/zero-one-ai/Yi-34B","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/zero-one-ai/Yi-6B":{"mode":"chat","base_model":"together/zero-one-ai/Yi-6B","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/EleutherAI/llemma_7b":{"mode":"chat","base_model":"together/EleutherAI/llemma_7b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/huggyllama/llama-65b":{"mode":"chat","base_model":"together/huggyllama/llama-65b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/llama-2-13b":{"mode":"chat","base_model":"together/togethercomputer/llama-2-13b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/llama-2-70b":{"mode":"chat","base_model":"together/togethercomputer/llama-2-70b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/llama-2-7b":{"mode":"chat","base_model":"together/togethercomputer/llama-2-7b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/microsoft/phi-2":{"mode":"chat","base_model":"together/microsoft/phi-2","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Nexusflow/NexusRaven-V2-13B":{"mode":"chat","base_model":"together/Nexusflow/NexusRaven-V2-13B","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/Qwen-7B":{"mode":"chat","base_model":"together/togethercomputer/Qwen-7B","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/falcon-40b":{"mode":"chat","base_model":"together/togethercomputer/falcon-40b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/falcon-7b":{"mode":"chat","base_model":"together/togethercomputer/falcon-7b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/GPT-JT-6B-v1":{"mode":"chat","base_model":"together/togethercomputer/GPT-JT-6B-v1","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/GPT-JT-Moderation-6B":{"mode":"chat","base_model":"together/togethercomputer/GPT-JT-Moderation-6B","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/LLaMA-2-7B-32K":{"mode":"chat","base_model":"together/togethercomputer/LLaMA-2-7B-32K","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/RedPajama-INCITE-Base-3B-v1":{"mode":"chat","base_model":"together/togethercomputer/RedPajama-INCITE-Base-3B-v1","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/RedPajama-INCITE-7B-Base":{"mode":"chat","base_model":"together/togethercomputer/RedPajama-INCITE-7B-Base","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/RedPajama-INCITE-Instruct-3B-v1":{"mode":"chat","base_model":"together/togethercomputer/RedPajama-INCITE-Instruct-3B-v1","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/RedPajama-INCITE-7B-Instruct":{"mode":"chat","base_model":"together/togethercomputer/RedPajama-INCITE-7B-Instruct","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/StripedHyena-Hessian-7B":{"mode":"chat","base_model":"together/togethercomputer/StripedHyena-Hessian-7B","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/WizardLM/WizardLM-70B-V1.0":{"mode":"chat","base_model":"together/WizardLM/WizardLM-70B-V1.0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/mistralai/Mistral-7B-v0.1":{"mode":"chat","base_model":"together/mistralai/Mistral-7B-v0.1","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/mistralai/Mixtral-8x7B-v0.1":{"mode":"chat","base_model":"together/mistralai/Mixtral-8x7B-v0.1","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-13b-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-13b-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-34b-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-34b-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-70b-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-70b-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-7b-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-7b-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-13b-Python-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-13b-Python-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-34b-Python-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-34b-Python-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-70b-Python-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-70b-Python-hf","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/codellama/CodeLlama-7b-Python-hf":{"mode":"chat","base_model":"together/codellama/CodeLlama-7b-Python-hf","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NumbersStation/nsql-llama-2-7B":{"mode":"chat","base_model":"together/NumbersStation/nsql-llama-2-7B","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Phind/Phind-CodeLlama-34B-Python-v1":{"mode":"chat","base_model":"together/Phind/Phind-CodeLlama-34B-Python-v1","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Phind/Phind-CodeLlama-34B-v2":{"mode":"chat","base_model":"together/Phind/Phind-CodeLlama-34B-v2","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/WizardLM/WizardCoder-Python-34B-V1.0":{"mode":"chat","base_model":"together/WizardLM/WizardCoder-Python-34B-V1.0","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/WizardLM/WizardCoder-15B-V1.0":{"mode":"chat","base_model":"together/WizardLM/WizardCoder-15B-V1.0","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/prompthero/openjourney":{"mode":"image_generation","base_model":"together/prompthero/openjourney","provider":"litellm","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/runwayml/stable-diffusion-v1-5":{"mode":"image_generation","base_model":"together/runwayml/stable-diffusion-v1-5","provider":"litellm","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/SG161222/Realistic_Vision_V3.0_VAE":{"mode":"image_generation","base_model":"together/SG161222/Realistic_Vision_V3.0_VAE","provider":"litellm","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/stabilityai/stable-diffusion-2-1":{"mode":"image_generation","base_model":"together/stabilityai/stable-diffusion-2-1","provider":"litellm","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/stabilityai/stable-diffusion-xl-base-1.0":{"mode":"image_generation","base_model":"together/stabilityai/stable-diffusion-xl-base-1.0","provider":"litellm","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/wavymulder/Analog-Diffusion":{"mode":"image_generation","base_model":"together/wavymulder/Analog-Diffusion","provider":"litellm","model_parameters":[{"id":"steps","label":"Steps","helpText":"Sampling steps is the number of iterations that the model runs to go from random noise to a recognizable image based on the text.","type":"number","default":20,"range":{"min":1,"max":100}},{"id":"seed","label":"Seed","helpText":"Random seed used for generation.","type":"number","default":42,"range":{"min":1,"max":10000}},{"id":"results","label":"Results","helpText":"Number of result images to generate.","type":"number","default":1,"range":{"min":1,"max":12}},{"id":"height","label":"Height","helpText":"Height of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"width","label":"Width","helpText":"Width of the image to generate in number of pixels.","type":"number","default":256,"range":{"min":256,"max":1024,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Meta-Llama/Llama-Guard-7b":{"mode":"moderation","base_model":"together/Meta-Llama/Llama-Guard-7b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"together/databricks/dolly-v2-12b":{"mode":"chat","base_model":"together/databricks/dolly-v2-12b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/databricks/dolly-v2-3b":{"mode":"chat","base_model":"together/databricks/dolly-v2-3b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/databricks/dolly-v2-7b":{"mode":"chat","base_model":"together/databricks/dolly-v2-7b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/HuggingFaceH4/zephyr-7b-beta":{"mode":"chat","base_model":"together/HuggingFaceH4/zephyr-7b-beta","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/HuggingFaceH4/starchat-alpha":{"mode":"chat","base_model":"together/HuggingFaceH4/starchat-alpha","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/OpenAssistant/oasst-sft-4-pythia-12b-epoch-3.5":{"mode":"chat","base_model":"together/OpenAssistant/oasst-sft-4-pythia-12b-epoch-3.5","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/OpenAssistant/stablelm-7b-sft-v7-epoch-3":{"mode":"chat","base_model":"together/OpenAssistant/stablelm-7b-sft-v7-epoch-3","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/Koala-13B":{"mode":"chat","base_model":"together/togethercomputer/Koala-13B","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/Koala-7B":{"mode":"chat","base_model":"together/togethercomputer/Koala-7B","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/lmsys/vicuna-13b-v1.3":{"mode":"chat","base_model":"together/lmsys/vicuna-13b-v1.3","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/lmsys/vicuna-7b-v1.3":{"mode":"chat","base_model":"together/lmsys/vicuna-7b-v1.3","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/lmsys/fastchat-t5-3b-v1.0":{"mode":"chat","base_model":"together/lmsys/fastchat-t5-3b-v1.0","provider":"litellm","max_input_tokens":512,"max_output_tokens":512,"max_tokens":512,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/mpt-30b-chat":{"mode":"chat","base_model":"together/togethercomputer/mpt-30b-chat","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/mpt-7b-chat":{"mode":"chat","base_model":"together/togethercomputer/mpt-7b-chat","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/guanaco-13b":{"mode":"chat","base_model":"together/togethercomputer/guanaco-13b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/guanaco-33b":{"mode":"chat","base_model":"together/togethercomputer/guanaco-33b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/guanaco-65b":{"mode":"chat","base_model":"together/togethercomputer/guanaco-65b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/togethercomputer/guanaco-7b":{"mode":"chat","base_model":"together/togethercomputer/guanaco-7b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/defog/sqlcoder":{"mode":"chat","base_model":"together/defog/sqlcoder","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/EleutherAI/gpt-j-6b":{"mode":"chat","base_model":"together/EleutherAI/gpt-j-6b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/EleutherAI/gpt-neox-20b":{"mode":"chat","base_model":"together/EleutherAI/gpt-neox-20b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/EleutherAI/pythia-12b-v0":{"mode":"chat","base_model":"together/EleutherAI/pythia-12b-v0","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/EleutherAI/pythia-1b-v0":{"mode":"chat","base_model":"together/EleutherAI/pythia-1b-v0","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/EleutherAI/pythia-2.8b-v0":{"mode":"chat","base_model":"together/EleutherAI/pythia-2.8b-v0","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/EleutherAI/pythia-6.9b":{"mode":"chat","base_model":"together/EleutherAI/pythia-6.9b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/google/flan-t5-xl":{"mode":"chat","base_model":"together/google/flan-t5-xl","provider":"litellm","max_input_tokens":512,"max_output_tokens":512,"max_tokens":512,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/google/flan-t5-xxl":{"mode":"chat","base_model":"together/google/flan-t5-xxl","provider":"litellm","max_input_tokens":512,"max_output_tokens":512,"max_tokens":512,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":512}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/huggyllama/llama-13b":{"mode":"chat","base_model":"together/huggyllama/llama-13b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/huggyllama/llama-30b":{"mode":"chat","base_model":"together/huggyllama/llama-30b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/huggyllama/llama-7b":{"mode":"chat","base_model":"together/huggyllama/llama-7b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/mosaicml/mpt-7b":{"mode":"chat","base_model":"together/mosaicml/mpt-7b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/mosaicml/mpt-7b-instruct":{"mode":"chat","base_model":"together/mosaicml/mpt-7b-instruct","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Hermes-13b":{"mode":"chat","base_model":"together/NousResearch/Nous-Hermes-13b","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NumbersStation/nsql-6B":{"mode":"chat","base_model":"together/NumbersStation/nsql-6B","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/stabilityai/stablelm-base-alpha-3b":{"mode":"chat","base_model":"together/stabilityai/stablelm-base-alpha-3b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/stabilityai/stablelm-base-alpha-7b":{"mode":"chat","base_model":"together/stabilityai/stablelm-base-alpha-7b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/bigcode/starcoder":{"mode":"chat","base_model":"together/bigcode/starcoder","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Replit-Code-v1 (3B)":{"mode":"chat","base_model":"together/Replit-Code-v1 (3B)","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Salesforce/codegen2-16B":{"mode":"chat","base_model":"together/Salesforce/codegen2-16B","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/Salesforce/codegen2-7B":{"mode":"chat","base_model":"together/Salesforce/codegen2-7B","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2048}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"anthropic/claude-instant-1.2":{"mode":"chat","base_model":"anthropic/claude-instant-1.2","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"source":"merged_from_llm_models_csv"},"anthropic/claude-2.1":{"mode":"chat","base_model":"anthropic/claude-2.1","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"source":"merged_from_llm_models_csv"},"anthropic/claude-3-opus-20240229":{"mode":"chat","base_model":"anthropic/claude-3-opus-20240229","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3-sonnet-20240229":{"mode":"image_generation","base_model":"anthropic/claude-3-sonnet-20240229","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3-5-sonnet-20241022":{"mode":"image_generation","base_model":"anthropic/claude-3-5-sonnet-20241022","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3-5-sonnet-latest":{"mode":"image_generation","base_model":"anthropic/claude-3-5-sonnet-latest","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3-haiku-20240307":{"mode":"image_generation","base_model":"anthropic/claude-3-haiku-20240307","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-3-7-sonnet-latest":{"mode":"image_generation","base_model":"anthropic/claude-3-7-sonnet-latest","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-sonnet-4-20250514":{"mode":"image_generation","base_model":"anthropic/claude-sonnet-4-20250514","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-opus-4-20250514":{"mode":"image_generation","base_model":"anthropic/claude-opus-4-20250514","provider":"litellm","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-opus-4-1-20250805":{"mode":"image_generation","base_model":"anthropic/claude-opus-4-1-20250805","provider":"litellm","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-sonnet-4-5-20250929":{"mode":"image_generation","base_model":"anthropic/claude-sonnet-4-5-20250929","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-opus-4-5":{"mode":"image_generation","base_model":"anthropic/claude-opus-4-5","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1},"disabledCondition":{"paramId":"thinking","operator":"eq","value":true,"setValue":1,"disabledText":"Temperature is disabled when extended thinking is enabled."}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"effort","label":"Effort","helpText":"Set the effort level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"thinking","label":"Extended thinking","helpText":"Enable the extended thinking parameter","type":"boolean","default":true},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"anthropic/claude-opus-4-5-20251101":{"mode":"image_generation","base_model":"anthropic/claude-opus-4-5-20251101","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1},"disabledCondition":{"paramId":"thinking","operator":"eq","value":true,"setValue":1,"disabledText":"Temperature is disabled when extended thinking is enabled."}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"effort","label":"Effort","helpText":"Set the effort level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"thinking","label":"Extended thinking","helpText":"Enable the extended thinking parameter","type":"boolean","default":true},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"azure/gpt-4-turbo-vision-128k":{"mode":"image_generation","base_model":"azure/gpt-4-turbo-vision-128k","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"azure/gpt-4o-audio-preview":{"mode":"chat","base_model":"azure/gpt-4o-audio-preview","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"azure/gpt-4o-audio-preview-2025-06-03":{"mode":"chat","base_model":"azure/gpt-4o-audio-preview-2025-06-03","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"azure/gpt-4o-audio-preview-2024-10-01":{"mode":"chat","base_model":"azure/gpt-4o-audio-preview-2024-10-01","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"azure/gpt-4o-mini-audio-preview":{"mode":"chat","base_model":"azure/gpt-4o-mini-audio-preview","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"modalities","label":"Modalities","helpText":"The modalities to use for the model.","type":"select","multiple":true,"hidden":true,"default":["text"],"options":[{"label":"Text","value":"text"},{"label":"Audio","value":"audio"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_vision":true,"source":"merged_from_llm_models_csv"},"azure/phi-4":{"mode":"chat","base_model":"azure/phi-4","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192},{"id":"imageDetail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"presencePenalty","label":"Presence penalty","helpText":"How much to penalize new tokens based on whether they appear in the text so far. Increases the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":16384}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"azure/gpt-5.1-chat-latest":{"mode":"image_generation","base_model":"azure/gpt-5.1-chat-latest","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01},"disabled":true,"disabledText":"Temperature value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01},"disabled":true,"disabledText":"Top P value is fixed at 1 and cannot be changed at the moment for this model."},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0,"disabled":true,"disabledText":"Frequency Penalty value is fixed at 0 and cannot be changed at the moment for this model."},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":128000,"range":{"min":1,"max":128000}}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"azure/DeepSeek-V3.1":{"mode":"chat","base_model":"azure/DeepSeek-V3.1","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"azure/DeepSeek-R1-0528":{"mode":"chat","base_model":"azure/DeepSeek-R1-0528","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"azure/DeepSeek-V3-0324":{"mode":"chat","base_model":"azure/DeepSeek-V3-0324","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"azure/DeepSeek-V3":{"mode":"chat","base_model":"azure/DeepSeek-V3","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"azure/DeepSeek-R1":{"mode":"chat","base_model":"azure/DeepSeek-R1","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"responseFormat","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequencyPenalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-pro":{"mode":"image_generation","base_model":"google/gemini-1.5-pro","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-pro-latest":{"mode":"image_generation","base_model":"google/gemini-1.5-pro-latest","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-pro-002":{"mode":"image_generation","base_model":"google/gemini-1.5-pro-002","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-flash":{"mode":"image_generation","base_model":"google/gemini-1.5-flash","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-flash-latest":{"mode":"image_generation","base_model":"google/gemini-1.5-flash-latest","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-flash-002":{"mode":"image_generation","base_model":"google/gemini-1.5-flash-002","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-flash-8b":{"mode":"image_generation","base_model":"google/gemini-1.5-flash-8b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-flash-8b-latest":{"mode":"image_generation","base_model":"google/gemini-1.5-flash-8b-latest","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-1.5-flash-8b-001":{"mode":"image_generation","base_model":"google/gemini-1.5-flash-8b-001","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash-lite-preview-02-05":{"mode":"image_generation","base_model":"google/gemini-2.0-flash-lite-preview-02-05","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-pro-preview-03-25":{"mode":"image_generation","base_model":"google/gemini-2.5-pro-preview-03-25","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash":{"mode":"chat","base_model":"google/gemini-2.0-flash","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-pro":{"mode":"chat","base_model":"google/gemini-2.5-pro","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-flash-preview-05-20":{"mode":"chat","base_model":"google/gemini-2.5-flash-preview-05-20","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-flash-preview-04-17":{"mode":"chat","base_model":"google/gemini-2.5-flash-preview-04-17","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.5-flash":{"mode":"chat","base_model":"google/gemini-2.5-flash","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash-exp":{"mode":"chat","base_model":"google/gemini-2.0-flash-exp","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash-001":{"mode":"chat","base_model":"google/gemini-2.0-flash-001","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash-lite":{"mode":"image_generation","base_model":"google/gemini-2.0-flash-lite","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash-lite-preview":{"mode":"image_generation","base_model":"google/gemini-2.0-flash-lite-preview","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-pro-exp":{"mode":"image_generation","base_model":"google/gemini-2.0-pro-exp","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-pro-exp-02-05":{"mode":"image_generation","base_model":"google/gemini-2.0-pro-exp-02-05","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"google/gemini-3-pro-preview":{"mode":"image_generation","base_model":"google/gemini-3-pro-preview","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call. Supports Google Search, File Search, Code Execution, URL Context, and Function Calling.","type":"select"},{"id":"media_resolution","label":"Media Resolution","helpText":"Higher resolutions may provide better understanding but use more tokens.","type":"select","default":"media_resolution_medium","options":[{"label":"Low","value":"media_resolution_low"},{"label":"Medium","value":"media_resolution_medium"},{"label":"High","value":"media_resolution_high"}]},{"id":"thinking_level","label":"Thinking Level","helpText":"Set the thinking level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"temperature","label":"Temperature","helpText":"For Gemini 3, best results at default 1.0. Lower values may impact reasoning.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Output length","helpText":"Maximum number of tokens in response","type":"number","default":32768,"range":{"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":5,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"google/gemini-2.0-flash-thinking-exp-01-21":{"mode":"image_generation","base_model":"google/gemini-2.0-flash-thinking-exp-01-21","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":10,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"topK","label":"Top K","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":1,"max":100,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"groq/mixtral-8x7b":{"mode":"chat","base_model":"groq/mixtral-8x7b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"groq/llama3-8b-8192":{"mode":"chat","base_model":"groq/llama3-8b-8192","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"groq/llama3-70b-8192":{"mode":"chat","base_model":"groq/llama3-70b-8192","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"groq/llama-3.1-70b-versatile":{"mode":"chat","base_model":"groq/llama-3.1-70b-versatile","provider":"litellm","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"groq/llama-3.1-405b-reasoning":{"mode":"chat","base_model":"groq/llama-3.1-405b-reasoning","provider":"litellm","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"groq/deepseek-r1-distill-llama-70b":{"mode":"chat","base_model":"groq/deepseek-r1-distill-llama-70b","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"together/Qwen/Qwen1.5-72B-Chat":{"mode":"chat","base_model":"together/Qwen/Qwen1.5-72B-Chat","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/NousResearch/Nous-Hermes-2-Mistral-7B-DPO":{"mode":"chat","base_model":"together/NousResearch/Nous-Hermes-2-Mistral-7B-DPO","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/deepseek-ai/deepseek-coder-33b-instruct":{"mode":"chat","base_model":"together/deepseek-ai/deepseek-coder-33b-instruct","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/deepseek-ai/DeepSeek-R1":{"mode":"chat","base_model":"together/deepseek-ai/DeepSeek-R1","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"together/google/gemma-7b-it":{"mode":"chat","base_model":"together/google/gemma-7b-it","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/meta-llama/Llama-3.2-11B-Vision-Instruct-Turbo":{"mode":"chat","base_model":"together/meta-llama/Llama-3.2-11B-Vision-Instruct-Turbo","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/meta-llama/Llama-3-8b-chat-hf":{"mode":"chat","base_model":"together/meta-llama/Llama-3-8b-chat-hf","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"together/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"mode":"image_generation","base_model":"together/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","provider":"litellm","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"min_p","label":"Min P","helpText":"A number between 0 and 1 that can be used as an alternative to top_p and top-k","type":"number","default":0,"range":{"min":0,"max":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}],"source":"merged_from_llm_models_csv"},"together/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"mode":"image_generation","base_model":"together/meta-llama/Llama-4-Scout-17B-16E-Instruct","provider":"litellm","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"min_p","label":"Min P","helpText":"A number between 0 and 1 that can be used as an alternative to top_p and top-k","type":"number","default":0,"range":{"min":0,"max":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}],"source":"merged_from_llm_models_csv"},"together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo":{"mode":"chat","base_model":"together/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo":{"mode":"chat","base_model":"together/meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo":{"mode":"chat","base_model":"together/meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"together/meta-llama/Llama-3.2-90B-Vision-Instruct-Turbo":{"mode":"image_generation","base_model":"together/meta-llama/Llama-3.2-90B-Vision-Instruct-Turbo","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"together/meta-llama/Llama-3.2-3B-Instruct-Turbo":{"mode":"chat","base_model":"together/meta-llama/Llama-3.2-3B-Instruct-Turbo","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cohere/command-nightly":{"mode":"chat","base_model":"cohere/command-nightly","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":128000,"step":10}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cohere/command-light":{"mode":"chat","base_model":"cohere/command-light","provider":"litellm","max_input_tokens":4000,"max_output_tokens":4000,"max_tokens":4000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4000}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":4000,"step":1}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"cohere/command-light-nightly":{"mode":"chat","base_model":"cohere/command-light-nightly","provider":"litellm","max_input_tokens":4000,"max_output_tokens":4000,"max_tokens":4000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.3,"range":{"min":0,"max":1,"step":0.1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":4000}},{"id":"max_input_tokens","label":"Max input tokens","helpText":"The maximum number of input tokens to send to the model. If not specified, max_input_tokens is the model's context length limit minus a small buffer.","type":"number","default":1000,"range":{"min":1,"max":4000,"step":1}},{"id":"p","label":"P","helpText":"Ensures that only the most likely tokens, with total probability mass of p, are considered for generation at each step. If both k and p are enabled, p acts after k.","type":"number","default":0.75,"range":{"min":0.01,"max":0.99,"step":0.01}},{"id":"k","label":"K","helpText":"Ensures only the top k most likely tokens are considered for generation at each step.","type":"number","default":10,"range":{"min":0,"max":500,"step":1}},{"id":"citation_quality","label":"Citation quality","helpText":"Dictates the approach taken to generating citations as part of the RAG flow by allowing the user to specify whether they want 'accurate' results or 'fast' results.","type":"select","options":[{"label":"Accurate","value":"accurate"},{"label":"Fast","value":"fast"}]},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Used to reduce repetitiveness of generated tokens. The higher the value, the stronger a penalty is applied to previously present tokens, proportional to how many times they have already appeared in the prompt or prior generation.","type":"number","range":{"min":0.1,"max":1,"step":0.1}},{"id":"seed","label":"Seed","helpText":"If specified, the backend will make a best effort to sample tokens deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":1000000}},{"id":"force_single_step","label":"Force single step","helpText":"Forces the chat to be single step.","type":"boolean","default":false},{"id":"preamble","label":"Preamble","helpText":"A short set of instructions given to an AI language model at the start of a conversation. It tells the AI how to behave and respond, like setting the rules of the game before playing. Users can change these instructions to make the AI act differently.","type":"text"},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-3-haiku-20240307-v1:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-3-haiku-20240307-v1:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-v2":{"mode":"chat","base_model":"bedrock/anthropic.claude-v2","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-instant-v1":{"mode":"chat","base_model":"bedrock/anthropic.claude-instant-v1","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-v2:1":{"mode":"chat","base_model":"bedrock/anthropic.claude-v2:1","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-3-sonnet-20240229-v1:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-3-sonnet-20240229-v1:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-3-opus-20240229-v1:0":{"mode":"chat","base_model":"bedrock/anthropic.claude-3-opus-20240229-v1:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-3-7-sonnet-20250219-v1:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-3-7-sonnet-20250219-v1:0","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-opus-4-20250514-v1:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-opus-4-20250514-v1:0","provider":"litellm","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-sonnet-4-20250514-v1:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-sonnet-4-20250514-v1:0","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-sonnet-4-5-20250929-v1:0","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"image_generation","base_model":"bedrock/anthropic.claude-haiku-4-5-20251001-v1:0","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/claude-opus-4-5":{"mode":"image_generation","base_model":"bedrock/claude-opus-4-5","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1},"disabledCondition":{"paramId":"thinking","operator":"eq","value":true,"setValue":1,"disabledText":"Temperature is disabled when extended thinking is enabled."}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"effort","label":"Effort","helpText":"Set the effort level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"thinking","label":"Extended thinking","helpText":"Enable the extended thinking parameter","type":"boolean","default":true},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/claude-opus-4-5-20251101":{"mode":"image_generation","base_model":"bedrock/claude-opus-4-5-20251101","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1},"disabledCondition":{"paramId":"thinking","operator":"eq","value":true,"setValue":1,"disabledText":"Temperature is disabled when extended thinking is enabled."}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"effort","label":"Effort","helpText":"Set the effort level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"thinking","label":"Extended thinking","helpText":"Enable the extended thinking parameter","type":"boolean","default":true},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/amazon.nova-lite-v1:0":{"mode":"image_generation","base_model":"bedrock/amazon.nova-lite-v1:0","provider":"litellm","max_input_tokens":5120,"max_output_tokens":5120,"max_tokens":5120,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":5120,"range":{"min":1,"max":5120}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/amazon.nova-micro-v1:0":{"mode":"chat","base_model":"bedrock/amazon.nova-micro-v1:0","provider":"litellm","max_input_tokens":5120,"max_output_tokens":5120,"max_tokens":5120,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":5120,"range":{"min":1,"max":5120}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/amazon.nova-pro-v1:0":{"mode":"image_generation","base_model":"bedrock/amazon.nova-pro-v1:0","provider":"litellm","max_input_tokens":5120,"max_output_tokens":5120,"max_tokens":5120,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":5120,"range":{"min":1,"max":5120}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"bedrock/amazon.titan-text-lite-v1":{"mode":"chat","base_model":"bedrock/amazon.titan-text-lite-v1","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"source":"merged_from_llm_models_csv"},"bedrock/amazon.titan-text-express-v1":{"mode":"chat","base_model":"bedrock/amazon.titan-text-express-v1","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}}],"source":"merged_from_llm_models_csv"},"bedrock/amazon.titan-text-premier-v1:0":{"mode":"chat","base_model":"bedrock/amazon.titan-text-premier-v1:0","provider":"litellm","max_input_tokens":3000,"max_output_tokens":3000,"max_tokens":3000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"stopSequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":3000,"range":{"min":1,"max":3000}}],"source":"merged_from_llm_models_csv"},"bedrock/meta.llama3-70b-instruct-v1:0":{"mode":"chat","base_model":"bedrock/meta.llama3-70b-instruct-v1:0","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}}],"source":"merged_from_llm_models_csv"},"bedrock/meta.llama3-8b-instruct-v1:0":{"mode":"chat","base_model":"bedrock/meta.llama3-8b-instruct-v1:0","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}}],"source":"merged_from_llm_models_csv"},"bedrock/meta.llama3-1-70b-instruct-v1:0":{"mode":"chat","base_model":"bedrock/meta.llama3-1-70b-instruct-v1:0","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}}],"source":"merged_from_llm_models_csv"},"bedrock/meta.llama3-1-8b-instruct-v1:0":{"mode":"chat","base_model":"bedrock/meta.llama3-1-8b-instruct-v1:0","provider":"litellm","max_input_tokens":2048,"max_output_tokens":2048,"max_tokens":2048,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}}],"source":"merged_from_llm_models_csv"},"bedrock/meta.llama3-2-90b-instruct-v1:0":{"mode":"image_generation","base_model":"bedrock/meta.llama3-2-90b-instruct-v1:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"source":"merged_from_llm_models_csv"},"bedrock/meta.llama3-2-11b-instruct-v1:0":{"mode":"image_generation","base_model":"bedrock/meta.llama3-2-11b-instruct-v1:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"source":"merged_from_llm_models_csv"},"bedrock/meta.llama3-2-3b-instruct-v1:0":{"mode":"chat","base_model":"bedrock/meta.llama3-2-3b-instruct-v1:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"source":"merged_from_llm_models_csv"},"bedrock/meta.llama3-2-1b-instruct-v1:0":{"mode":"chat","base_model":"bedrock/meta.llama3-2-1b-instruct-v1:0","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}],"source":"merged_from_llm_models_csv"},"bedrock/deepseek.r1-v1:0":{"mode":"chat","base_model":"bedrock/deepseek.r1-v1:0","source":"merged_from_llm_models_csv","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_reasoning":true,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}]},"bedrock/deepseek-llm-r1-distill-qwen-7b":{"mode":"chat","base_model":"bedrock/deepseek-llm-r1-distill-qwen-7b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/deepseek-llm-r1-distill-qwen-32b":{"mode":"chat","base_model":"bedrock/deepseek-llm-r1-distill-qwen-32b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/deepseek-llm-r1-distill-qwen-14b":{"mode":"chat","base_model":"bedrock/deepseek-llm-r1-distill-qwen-14b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/deepseek-llm-r1-distill-llama-8b":{"mode":"chat","base_model":"bedrock/deepseek-llm-r1-distill-llama-8b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/deepseek-llm-r1-distill-llama-70b":{"mode":"chat","base_model":"bedrock/deepseek-llm-r1-distill-llama-70b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/huggingface-llm-qwen2-5-coder-7b-instruct":{"mode":"chat","base_model":"bedrock/huggingface-llm-qwen2-5-coder-7b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"bedrock/huggingface-llm-qwen2-5-coder-32b-instruct":{"mode":"chat","base_model":"bedrock/huggingface-llm-qwen2-5-coder-32b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"bedrock/huggingface-llm-qwen2-5-7b-instruct":{"mode":"chat","base_model":"bedrock/huggingface-llm-qwen2-5-7b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"bedrock/huggingface-llm-qwen2-5-72b-instruct":{"mode":"chat","base_model":"bedrock/huggingface-llm-qwen2-5-72b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"bedrock/huggingface-llm-qwen2-5-32b-instruct":{"mode":"chat","base_model":"bedrock/huggingface-llm-qwen2-5-32b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"bedrock/huggingface-llm-qwen2-5-14b-instruct":{"mode":"chat","base_model":"bedrock/huggingface-llm-qwen2-5-14b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":32768}}],"source":"merged_from_llm_models_csv"},"bedrock/openai.gpt-oss-120b-1:0":{"mode":"chat","base_model":"bedrock/openai.gpt-oss-120b-1:0","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":8192}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"bedrock/openai.gpt-oss-20b-1:0":{"mode":"chat","base_model":"bedrock/openai.gpt-oss-20b-1:0","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","type":"select","default":"medium","options":[{"value":"low","label":"Low"},{"value":"medium","label":"Medium"},{"value":"high","label":"High"}]},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":8192}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"mistral/ministral-8b-latest":{"mode":"chat","base_model":"mistral/ministral-8b-latest","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"source":"merged_from_llm_models_csv","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131000,"range":{"min":1,"max":131000}}]},"mistral/pixtral-12b":{"mode":"image_generation","base_model":"mistral/pixtral-12b","provider":"litellm","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131000}},{"id":"image_detail","label":"Image Detail","helpText":"By controlling the detail parameter, you have control over how the model processes the image and generates its textual understanding. You can override this value at attachment level.","type":"select","accesorKey":"detail","default":{"detail":"auto"},"options":[{"label":"Auto","value":"auto"},{"label":"Low","value":"low"},{"label":"High","value":"high"}]},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"name":"","strict":true,"schema":{}}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"source":"merged_from_llm_models_csv"},"mistral/mistral-saba-latest":{"mode":"chat","base_model":"mistral/mistral-saba-latest","provider":"litellm","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32000,"range":{"min":1,"max":32000}}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"mistral/ministral-3b-latest":{"mode":"chat","base_model":"mistral/ministral-3b-latest","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"source":"merged_from_llm_models_csv","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"maxTokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536},{"id":"topP","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0},"default":0.9},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131000,"range":{"min":1,"max":131000}}]},"fireworks_ai/accounts/yi-01-ai/models/yi-large":{"mode":"chat","base_model":"fireworks/accounts/yi-01-ai/models/yi-large","provider":"litellm","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/sentientfoundation/models/dobby-unhinged-llama-3-3-70b-new":{"mode":"chat","base_model":"fireworks/accounts/sentientfoundation/models/dobby-unhinged-llama-3-3-70b-new","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/alpha":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/alpha","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"fireworks_ai/accounts/fireworks/models/moa":{"mode":"chat","base_model":"fireworks/accounts/fireworks/models/moa","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/openrouter/horizon-beta":{"mode":"image_generation","base_model":"openrouter/openrouter/horizon-beta","provider":"litellm","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/codestral-2508":{"mode":"chat","base_model":"openrouter/mistralai/codestral-2508","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-30b-a3b-instruct-2507":{"mode":"chat","base_model":"openrouter/qwen/qwen3-30b-a3b-instruct-2507","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/z-ai/glm-4.5":{"mode":"chat","base_model":"openrouter/z-ai/glm-4.5","deprecation_date":"2026-12-31","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/z-ai/glm-4.5-air:free":{"mode":"chat","base_model":"openrouter/z-ai/glm-4.5-air:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/z-ai/glm-4.5-air":{"mode":"chat","base_model":"openrouter/z-ai/glm-4.5-air","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/z-ai/glm-4-32b":{"mode":"chat","base_model":"openrouter/z-ai/glm-4-32b","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/google/gemini-2.5-flash-lite":{"mode":"image_generation","base_model":"openrouter/google/gemini-2.5-flash-lite","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"merged_from_llm_models_csv","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"deprecation_date":"2026-10-20","supports_prompt_caching":true,"supports_web_search":true,"supports_video_input":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/moonshotai/kimi-k2:free":{"mode":"chat","base_model":"openrouter/moonshotai/kimi-k2:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/moonshotai/kimi-k2":{"mode":"chat","base_model":"openrouter/moonshotai/kimi-k2","max_input_tokens":63000,"max_output_tokens":63000,"max_tokens":63000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":63000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/thudm/glm-4.1v-9b-thinking":{"mode":"image_generation","base_model":"openrouter/thudm/glm-4.1v-9b-thinking","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/devstral-medium":{"mode":"chat","base_model":"openrouter/mistralai/devstral-medium","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/devstral-small":{"mode":"chat","base_model":"openrouter/mistralai/devstral-small","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/cognitivecomputations/dolphin-mistral-24b-venice-edition:free":{"mode":"chat","base_model":"openrouter/cognitivecomputations/dolphin-mistral-24b-venice-edition:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3n-e2b-it:free":{"mode":"chat","base_model":"openrouter/google/gemma-3n-e2b-it:free","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/tencent/hunyuan-a13b-instruct:free":{"mode":"chat","base_model":"openrouter/tencent/hunyuan-a13b-instruct:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/tencent/hunyuan-a13b-instruct":{"mode":"chat","base_model":"openrouter/tencent/hunyuan-a13b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/tngtech/deepseek-r1t2-chimera:free":{"mode":"chat","base_model":"openrouter/tngtech/deepseek-r1t2-chimera:free","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/morph/morph-v3-large":{"mode":"chat","base_model":"openrouter/morph/morph-v3-large","max_input_tokens":81920,"max_output_tokens":81920,"max_tokens":81920,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":81920}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/morph/morph-v3-fast":{"mode":"chat","base_model":"openrouter/morph/morph-v3-fast","max_input_tokens":81920,"max_output_tokens":81920,"max_tokens":81920,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":81920}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/baidu/ernie-4.5-300b-a47b":{"mode":"chat","base_model":"openrouter/baidu/ernie-4.5-300b-a47b","provider":"litellm","max_input_tokens":123000,"max_output_tokens":123000,"max_tokens":123000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":123000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/thedrummer/anubis-70b-v1.1":{"mode":"chat","base_model":"openrouter/thedrummer/anubis-70b-v1.1","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/inception/mercury":{"mode":"chat","base_model":"openrouter/inception/mercury","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-small-3.2-24b-instruct:free":{"mode":"image_generation","base_model":"openrouter/mistralai/mistral-small-3.2-24b-instruct:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/minimax/minimax-m1":{"mode":"chat","base_model":"openrouter/minimax/minimax-m1","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemini-2.5-flash-lite-preview-06-17":{"mode":"image_generation","base_model":"openrouter/google/gemini-2.5-flash-lite-preview-06-17","provider":"litellm","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/moonshotai/kimi-dev-72b:free":{"mode":"chat","base_model":"openrouter/moonshotai/kimi-dev-72b:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/o3-pro":{"mode":"image_generation","base_model":"openrouter/openai/o3-pro","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/x-ai/grok-3-mini":{"mode":"chat","base_model":"openrouter/x-ai/grok-3-mini","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/x-ai/grok-3":{"mode":"chat","base_model":"openrouter/x-ai/grok-3","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/magistral-small-2506":{"mode":"chat","base_model":"openrouter/mistralai/magistral-small-2506","provider":"litellm","max_input_tokens":40000,"max_output_tokens":40000,"max_tokens":40000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/magistral-medium-2506":{"mode":"chat","base_model":"openrouter/mistralai/magistral-medium-2506","provider":"litellm","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/magistral-medium-2506:thinking":{"mode":"chat","base_model":"openrouter/mistralai/magistral-medium-2506:thinking","provider":"litellm","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/google/gemini-2.5-pro-preview":{"mode":"image_generation","base_model":"openrouter/google/gemini-2.5-pro-preview","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"merged_from_llm_models_csv","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-r1-distill-qwen-7b":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-distill-qwen-7b","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-0528-qwen3-8b:free":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-0528-qwen3-8b:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-0528-qwen3-8b":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-0528-qwen3-8b","provider":"litellm","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-0528:free":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-0528:free","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/sarvamai/sarvam-m:free":{"mode":"chat","base_model":"openrouter/sarvamai/sarvam-m:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/thedrummer/valkyrie-49b-v1":{"mode":"chat","base_model":"openrouter/thedrummer/valkyrie-49b-v1","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/devstral-small-2505:free":{"mode":"chat","base_model":"openrouter/mistralai/devstral-small-2505:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/devstral-small-2505":{"mode":"chat","base_model":"openrouter/mistralai/devstral-small-2505","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3n-e4b-it:free":{"mode":"chat","base_model":"openrouter/google/gemma-3n-e4b-it:free","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3n-e4b-it":{"mode":"chat","base_model":"openrouter/google/gemma-3n-e4b-it","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/codex-mini":{"mode":"image_generation","base_model":"openrouter/openai/codex-mini","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/nousresearch/deephermes-3-mistral-24b-preview":{"mode":"chat","base_model":"openrouter/nousresearch/deephermes-3-mistral-24b-preview","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-medium-3":{"mode":"image_generation","base_model":"openrouter/mistralai/mistral-medium-3","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemini-2.5-pro-preview-05-06":{"mode":"image_generation","base_model":"openrouter/google/gemini-2.5-pro-preview-05-06","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"merged_from_llm_models_csv","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/arcee-ai/spotlight":{"mode":"image_generation","base_model":"openrouter/arcee-ai/spotlight","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/arcee-ai/maestro-reasoning":{"mode":"chat","base_model":"openrouter/arcee-ai/maestro-reasoning","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/arcee-ai/virtuoso-large":{"mode":"chat","base_model":"openrouter/arcee-ai/virtuoso-large","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/arcee-ai/coder-large":{"mode":"chat","base_model":"openrouter/arcee-ai/coder-large","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/microsoft/phi-4-reasoning-plus":{"mode":"chat","base_model":"openrouter/microsoft/phi-4-reasoning-plus","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/inception/mercury-coder":{"mode":"chat","base_model":"openrouter/inception/mercury-coder","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen3-4b:free":{"mode":"chat","base_model":"openrouter/qwen/qwen3-4b:free","provider":"litellm","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/opengvlab/internvl3-14b":{"mode":"image_generation","base_model":"openrouter/opengvlab/internvl3-14b","provider":"litellm","max_input_tokens":12288,"max_output_tokens":12288,"max_tokens":12288,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":12288}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-prover-v2":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-prover-v2","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-guard-4-12b":{"mode":"image_generation","base_model":"openrouter/meta-llama/llama-guard-4-12b","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":false,"supports_response_schema":false,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-30b-a3b:free":{"mode":"chat","base_model":"openrouter/qwen/qwen3-30b-a3b:free","provider":"litellm","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen3-30b-a3b":{"mode":"chat","base_model":"openrouter/qwen/qwen3-30b-a3b","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-8b:free":{"mode":"chat","base_model":"openrouter/qwen/qwen3-8b:free","provider":"litellm","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen3-8b":{"mode":"chat","base_model":"openrouter/qwen/qwen3-8b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-14b:free":{"mode":"chat","base_model":"openrouter/qwen/qwen3-14b:free","provider":"litellm","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen3-14b":{"mode":"chat","base_model":"openrouter/qwen/qwen3-14b","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-32b":{"mode":"chat","base_model":"openrouter/qwen/qwen3-32b","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen3-235b-a22b:free":{"mode":"chat","base_model":"openrouter/qwen/qwen3-235b-a22b:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen3-235b-a22b":{"mode":"chat","base_model":"openrouter/qwen/qwen3-235b-a22b","max_input_tokens":40960,"max_output_tokens":40960,"max_tokens":40960,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":40960}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/tngtech/deepseek-r1t-chimera:free":{"mode":"chat","base_model":"openrouter/tngtech/deepseek-r1t-chimera:free","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/tngtech/deepseek-r1t-chimera":{"mode":"chat","base_model":"openrouter/tngtech/deepseek-r1t-chimera","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/microsoft/mai-ds-r1:free":{"mode":"chat","base_model":"openrouter/microsoft/mai-ds-r1:free","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/microsoft/mai-ds-r1":{"mode":"chat","base_model":"openrouter/microsoft/mai-ds-r1","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/thudm/glm-z1-32b:free":{"mode":"chat","base_model":"openrouter/thudm/glm-z1-32b:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/thudm/glm-4-32b":{"mode":"chat","base_model":"openrouter/thudm/glm-4-32b","provider":"litellm","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/o4-mini-high":{"mode":"image_generation","base_model":"openrouter/openai/o4-mini-high","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/o3":{"mode":"image_generation","base_model":"openrouter/openai/o3","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"merged_from_llm_models_csv","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/o4-mini":{"mode":"image_generation","base_model":"openrouter/openai/o4-mini","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"merged_from_llm_models_csv","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/shisa-ai/shisa-v2-llama3.3-70b:free":{"mode":"chat","base_model":"openrouter/shisa-ai/shisa-v2-llama3.3-70b:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/shisa-ai/shisa-v2-llama3.3-70b":{"mode":"chat","base_model":"openrouter/shisa-ai/shisa-v2-llama3.3-70b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/eleutherai/llemma_7b":{"mode":"chat","base_model":"openrouter/eleutherai/llemma_7b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/alfredpros/codellama-7b-instruct-solidity":{"mode":"chat","base_model":"openrouter/alfredpros/codellama-7b-instruct-solidity","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/arliai/qwq-32b-arliai-rpr-v1:free":{"mode":"chat","base_model":"openrouter/arliai/qwq-32b-arliai-rpr-v1:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/arliai/qwq-32b-arliai-rpr-v1":{"mode":"chat","base_model":"openrouter/arliai/qwq-32b-arliai-rpr-v1","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/agentica-org/deepcoder-14b-preview:free":{"mode":"chat","base_model":"openrouter/agentica-org/deepcoder-14b-preview:free","provider":"litellm","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/agentica-org/deepcoder-14b-preview":{"mode":"chat","base_model":"openrouter/agentica-org/deepcoder-14b-preview","provider":"litellm","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/moonshotai/kimi-vl-a3b-thinking:free":{"mode":"image_generation","base_model":"openrouter/moonshotai/kimi-vl-a3b-thinking:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/moonshotai/kimi-vl-a3b-thinking":{"mode":"image_generation","base_model":"openrouter/moonshotai/kimi-vl-a3b-thinking","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/x-ai/grok-3-mini-beta":{"mode":"chat","base_model":"openrouter/x-ai/grok-3-mini-beta","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/x-ai/grok-3-beta":{"mode":"chat","base_model":"openrouter/x-ai/grok-3-beta","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/nvidia/llama-3.3-nemotron-super-49b-v1":{"mode":"chat","base_model":"openrouter/nvidia/llama-3.3-nemotron-super-49b-v1","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/nvidia/llama-3.1-nemotron-ultra-253b-v1:free":{"mode":"chat","base_model":"openrouter/nvidia/llama-3.1-nemotron-ultra-253b-v1:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/nvidia/llama-3.1-nemotron-ultra-253b-v1":{"mode":"chat","base_model":"openrouter/nvidia/llama-3.1-nemotron-ultra-253b-v1","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-4-maverick":{"mode":"image_generation","base_model":"openrouter/meta-llama/llama-4-maverick","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/meta-llama/llama-4-scout":{"mode":"image_generation","base_model":"openrouter/meta-llama/llama-4-scout","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-v3-base":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-v3-base","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/scb10x/llama3.1-typhoon2-70b-instruct":{"mode":"chat","base_model":"openrouter/scb10x/llama3.1-typhoon2-70b-instruct","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemini-2.5-pro-exp-03-25":{"mode":"image_generation","base_model":"openrouter/google/gemini-2.5-pro-exp-03-25","provider":"litellm","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen2.5-vl-32b-instruct:free":{"mode":"image_generation","base_model":"openrouter/qwen/qwen2.5-vl-32b-instruct:free","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen2.5-vl-32b-instruct":{"mode":"image_generation","base_model":"openrouter/qwen/qwen2.5-vl-32b-instruct","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-chat-v3-0324:free":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-chat-v3-0324:free","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/featherless/qwerky-72b:free":{"mode":"chat","base_model":"openrouter/featherless/qwerky-72b:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/o1-pro":{"mode":"image_generation","base_model":"openrouter/openai/o1-pro","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mistral-small-3.1-24b-instruct:free":{"mode":"image_generation","base_model":"openrouter/mistralai/mistral-small-3.1-24b-instruct:free","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3-4b-it:free":{"mode":"image_generation","base_model":"openrouter/google/gemma-3-4b-it:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3-4b-it":{"mode":"image_generation","base_model":"openrouter/google/gemma-3-4b-it","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":false,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/ai21/jamba-1.6-large":{"mode":"chat","base_model":"openrouter/ai21/jamba-1.6-large","provider":"litellm","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/ai21/jamba-1.6-mini":{"mode":"chat","base_model":"openrouter/ai21/jamba-1.6-mini","provider":"litellm","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":256000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3-12b-it:free":{"mode":"image_generation","base_model":"openrouter/google/gemma-3-12b-it:free","provider":"litellm","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3-12b-it":{"mode":"image_generation","base_model":"openrouter/google/gemma-3-12b-it","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/cohere/command-a":{"mode":"chat","base_model":"openrouter/cohere/command-a","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4o-mini-search-preview":{"mode":"chat","base_model":"openrouter/openai/gpt-4o-mini-search-preview","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/gpt-4o-search-preview":{"mode":"chat","base_model":"openrouter/openai/gpt-4o-search-preview","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/rekaai/reka-flash-3:free":{"mode":"chat","base_model":"openrouter/rekaai/reka-flash-3:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3-27b-it:free":{"mode":"image_generation","base_model":"openrouter/google/gemma-3-27b-it:free","provider":"litellm","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-3-27b-it":{"mode":"image_generation","base_model":"openrouter/google/gemma-3-27b-it","max_input_tokens":96000,"max_output_tokens":96000,"max_tokens":96000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":96000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/thedrummer/anubis-pro-105b-v1":{"mode":"chat","base_model":"openrouter/thedrummer/anubis-pro-105b-v1","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/thedrummer/skyfall-36b-v2":{"mode":"chat","base_model":"openrouter/thedrummer/skyfall-36b-v2","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/microsoft/phi-4-multimodal-instruct":{"mode":"image_generation","base_model":"openrouter/microsoft/phi-4-multimodal-instruct","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/perplexity/sonar-reasoning-pro":{"mode":"image_generation","base_model":"openrouter/perplexity/sonar-reasoning-pro","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/perplexity/sonar-pro":{"mode":"image_generation","base_model":"openrouter/perplexity/sonar-pro","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/perplexity/sonar-deep-research":{"mode":"chat","base_model":"openrouter/perplexity/sonar-deep-research","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwq-32b:free":{"mode":"chat","base_model":"openrouter/qwen/qwq-32b:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwq-32b":{"mode":"chat","base_model":"openrouter/qwen/qwq-32b","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/nousresearch/deephermes-3-llama-3-8b-preview:free":{"mode":"chat","base_model":"openrouter/nousresearch/deephermes-3-llama-3-8b-preview:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/google/gemini-2.0-flash-lite-001":{"mode":"image_generation","base_model":"openrouter/google/gemini-2.0-flash-lite-001","provider":"litellm","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3.7-sonnet:thinking":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3.7-sonnet:thinking","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3.7-sonnet:beta":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3.7-sonnet:beta","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/perplexity/r1-1776":{"mode":"chat","base_model":"openrouter/perplexity/r1-1776","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-saba":{"mode":"chat","base_model":"openrouter/mistralai/mistral-saba","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/cognitivecomputations/dolphin3.0-r1-mistral-24b:free":{"mode":"chat","base_model":"openrouter/cognitivecomputations/dolphin3.0-r1-mistral-24b:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/cognitivecomputations/dolphin3.0-r1-mistral-24b":{"mode":"chat","base_model":"openrouter/cognitivecomputations/dolphin3.0-r1-mistral-24b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/cognitivecomputations/dolphin3.0-mistral-24b:free":{"mode":"chat","base_model":"openrouter/cognitivecomputations/dolphin3.0-mistral-24b:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/cognitivecomputations/dolphin3.0-mistral-24b":{"mode":"chat","base_model":"openrouter/cognitivecomputations/dolphin3.0-mistral-24b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-guard-3-8b":{"mode":"chat","base_model":"openrouter/meta-llama/llama-guard-3-8b","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-distill-llama-8b":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-distill-llama-8b","provider":"litellm","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/aion-labs/aion-1.0":{"mode":"chat","base_model":"openrouter/aion-labs/aion-1.0","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/aion-labs/aion-1.0-mini":{"mode":"chat","base_model":"openrouter/aion-labs/aion-1.0-mini","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/aion-labs/aion-rp-llama-3.1-8b":{"mode":"chat","base_model":"openrouter/aion-labs/aion-rp-llama-3.1-8b","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen-vl-max":{"mode":"image_generation","base_model":"openrouter/qwen/qwen-vl-max","provider":"litellm","max_input_tokens":7500,"max_output_tokens":7500,"max_tokens":7500,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":7500}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen-turbo":{"mode":"chat","base_model":"openrouter/qwen/qwen-turbo","provider":"litellm","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen2.5-vl-72b-instruct:free":{"mode":"image_generation","base_model":"openrouter/qwen/qwen2.5-vl-72b-instruct:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen2.5-vl-72b-instruct":{"mode":"image_generation","base_model":"openrouter/qwen/qwen2.5-vl-72b-instruct","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_tool_choice":false,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen-plus":{"mode":"chat","base_model":"openrouter/qwen/qwen-plus","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen-max":{"mode":"chat","base_model":"openrouter/qwen/qwen-max","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-distill-qwen-1.5b":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-distill-qwen-1.5b","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-small-24b-instruct-2501:free":{"mode":"chat","base_model":"openrouter/mistralai/mistral-small-24b-instruct-2501:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-small-24b-instruct-2501":{"mode":"chat","base_model":"openrouter/mistralai/mistral-small-24b-instruct-2501","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-r1-distill-qwen-32b":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-distill-qwen-32b","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-distill-qwen-14b:free":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-distill-qwen-14b:free","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-distill-qwen-14b":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-distill-qwen-14b","provider":"litellm","max_input_tokens":64000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":64000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/perplexity/sonar-reasoning":{"mode":"chat","base_model":"openrouter/perplexity/sonar-reasoning","provider":"litellm","max_input_tokens":127000,"max_output_tokens":127000,"max_tokens":127000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":127000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/perplexity/sonar":{"mode":"image_generation","base_model":"openrouter/perplexity/sonar","max_input_tokens":127072,"max_output_tokens":127072,"max_tokens":127072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":127072}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/liquid/lfm-7b":{"mode":"chat","base_model":"openrouter/liquid/lfm-7b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/liquid/lfm-3b":{"mode":"chat","base_model":"openrouter/liquid/lfm-3b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-distill-llama-70b:free":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-distill-llama-70b:free","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/deepseek/deepseek-r1-distill-llama-70b":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1-distill-llama-70b","deprecation_date":"2026-09-28","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/deepseek/deepseek-r1:free":{"mode":"chat","base_model":"openrouter/deepseek/deepseek-r1:free","provider":"litellm","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":163840}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/minimax/minimax-01":{"mode":"image_generation","base_model":"openrouter/minimax/minimax-01","max_input_tokens":1000192,"max_output_tokens":1000192,"max_tokens":1000192,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000192}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/codestral-2501":{"mode":"chat","base_model":"openrouter/mistralai/codestral-2501","provider":"litellm","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/microsoft/phi-4":{"mode":"chat","base_model":"openrouter/microsoft/phi-4","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/sao10k/l3.3-euryale-70b":{"mode":"chat","base_model":"openrouter/sao10k/l3.3-euryale-70b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/x-ai/grok-2-vision-1212":{"mode":"image_generation","base_model":"openrouter/x-ai/grok-2-vision-1212","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/x-ai/grok-2-1212":{"mode":"chat","base_model":"openrouter/x-ai/grok-2-1212","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/cohere/command-r7b-12-2024":{"mode":"chat","base_model":"openrouter/cohere/command-r7b-12-2024","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemini-2.0-flash-exp:free":{"mode":"image_generation","base_model":"openrouter/google/gemini-2.0-flash-exp:free","provider":"litellm","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1048576}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.3-70b-instruct:free":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.3-70b-instruct:free","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.3-70b-instruct":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.3-70b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/amazon/nova-lite-v1":{"mode":"image_generation","base_model":"openrouter/amazon/nova-lite-v1","max_input_tokens":300000,"max_output_tokens":300000,"max_tokens":300000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":300000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/amazon/nova-micro-v1":{"mode":"chat","base_model":"openrouter/amazon/nova-micro-v1","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/amazon/nova-pro-v1":{"mode":"image_generation","base_model":"openrouter/amazon/nova-pro-v1","max_input_tokens":300000,"max_output_tokens":300000,"max_tokens":300000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":300000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwq-32b-preview":{"mode":"chat","base_model":"openrouter/qwen/qwq-32b-preview","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/gpt-4o-2024-11-20":{"mode":"image_generation","base_model":"openrouter/openai/gpt-4o-2024-11-20","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mistral-large-2411":{"mode":"chat","base_model":"openrouter/mistralai/mistral-large-2411","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-large-2407":{"mode":"chat","base_model":"openrouter/mistralai/mistral-large-2407","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/pixtral-large-2411":{"mode":"image_generation","base_model":"openrouter/mistralai/pixtral-large-2411","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/x-ai/grok-vision-beta":{"mode":"image_generation","base_model":"openrouter/x-ai/grok-vision-beta","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/infermatic/mn-inferor-12b":{"mode":"chat","base_model":"openrouter/infermatic/mn-inferor-12b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen-2.5-coder-32b-instruct:free":{"mode":"chat","base_model":"openrouter/qwen/qwen-2.5-coder-32b-instruct:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/raifle/sorcererlm-8x22b":{"mode":"chat","base_model":"openrouter/raifle/sorcererlm-8x22b","provider":"litellm","max_input_tokens":16000,"max_output_tokens":16000,"max_tokens":16000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/thedrummer/unslopnemo-12b":{"mode":"chat","base_model":"openrouter/thedrummer/unslopnemo-12b","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-3.5-haiku:beta":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3.5-haiku:beta","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3.5-haiku":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3.5-haiku","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3.5-haiku-20241022":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3.5-haiku-20241022","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/anthracite-org/magnum-v4-72b":{"mode":"chat","base_model":"openrouter/anthracite-org/magnum-v4-72b","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/anthropic/claude-3.5-sonnet:beta":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3.5-sonnet:beta","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/ministral-8b":{"mode":"chat","base_model":"openrouter/mistralai/ministral-8b","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/ministral-3b":{"mode":"chat","base_model":"openrouter/mistralai/ministral-3b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen-2.5-7b-instruct":{"mode":"chat","base_model":"openrouter/qwen/qwen-2.5-7b-instruct","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/nvidia/llama-3.1-nemotron-70b-instruct":{"mode":"chat","base_model":"openrouter/nvidia/llama-3.1-nemotron-70b-instruct","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/inflection/inflection-3-productivity":{"mode":"chat","base_model":"openrouter/inflection/inflection-3-productivity","provider":"litellm","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/inflection/inflection-3-pi":{"mode":"chat","base_model":"openrouter/inflection/inflection-3-pi","provider":"litellm","max_input_tokens":8000,"max_output_tokens":8000,"max_tokens":8000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemini-flash-1.5-8b":{"mode":"image_generation","base_model":"openrouter/google/gemini-flash-1.5-8b","provider":"litellm","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/thedrummer/rocinante-12b":{"mode":"chat","base_model":"openrouter/thedrummer/rocinante-12b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/liquid/lfm-40b":{"mode":"chat","base_model":"openrouter/liquid/lfm-40b","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/anthracite-org/magnum-v2-72b":{"mode":"chat","base_model":"openrouter/anthracite-org/magnum-v2-72b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.2-3b-instruct:free":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.2-3b-instruct:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.2-3b-instruct":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.2-3b-instruct","max_input_tokens":20000,"max_output_tokens":20000,"max_tokens":20000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":20000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/meta-llama/llama-3.2-90b-vision-instruct":{"mode":"image_generation","base_model":"openrouter/meta-llama/llama-3.2-90b-vision-instruct","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.2-1b-instruct":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.2-1b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/meta-llama/llama-3.2-11b-vision-instruct:free":{"mode":"image_generation","base_model":"openrouter/meta-llama/llama-3.2-11b-vision-instruct:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.2-11b-vision-instruct":{"mode":"image_generation","base_model":"openrouter/meta-llama/llama-3.2-11b-vision-instruct","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen-2.5-72b-instruct:free":{"mode":"chat","base_model":"openrouter/qwen/qwen-2.5-72b-instruct:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen-2.5-72b-instruct":{"mode":"chat","base_model":"openrouter/qwen/qwen-2.5-72b-instruct","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/neversleep/llama-3.1-lumimaid-8b":{"mode":"chat","base_model":"openrouter/neversleep/llama-3.1-lumimaid-8b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/o1-mini-2024-09-12":{"mode":"chat","base_model":"openrouter/openai/o1-mini-2024-09-12","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/openai/o1-mini":{"mode":"chat","base_model":"openrouter/openai/o1-mini","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/pixtral-12b":{"mode":"image_generation","base_model":"openrouter/mistralai/pixtral-12b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/cohere/command-r-08-2024":{"mode":"chat","base_model":"openrouter/cohere/command-r-08-2024","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/cohere/command-r-plus-08-2024":{"mode":"chat","base_model":"openrouter/cohere/command-r-plus-08-2024","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/qwen/qwen-2.5-vl-7b-instruct":{"mode":"image_generation","base_model":"openrouter/qwen/qwen-2.5-vl-7b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/sao10k/l3.1-euryale-70b":{"mode":"chat","base_model":"openrouter/sao10k/l3.1-euryale-70b","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/microsoft/phi-3.5-mini-128k-instruct":{"mode":"chat","base_model":"openrouter/microsoft/phi-3.5-mini-128k-instruct","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/nousresearch/hermes-3-llama-3.1-70b":{"mode":"chat","base_model":"openrouter/nousresearch/hermes-3-llama-3.1-70b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/nousresearch/hermes-3-llama-3.1-405b":{"mode":"chat","base_model":"openrouter/nousresearch/hermes-3-llama-3.1-405b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/chatgpt-4o-latest":{"mode":"image_generation","base_model":"openrouter/openai/chatgpt-4o-latest","provider":"litellm","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/sao10k/l3-lunaris-8b":{"mode":"chat","base_model":"openrouter/sao10k/l3-lunaris-8b","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4o-2024-08-06":{"mode":"image_generation","base_model":"openrouter/openai/gpt-4o-2024-08-06","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/meta-llama/llama-3.1-405b":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.1-405b","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.1-70b-instruct":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.1-70b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/meta-llama/llama-3.1-405b-instruct:free":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.1-405b-instruct:free","provider":"litellm","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.1-405b-instruct":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.1-405b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3.1-8b-instruct":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3.1-8b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/mistralai/mistral-nemo:free":{"mode":"chat","base_model":"openrouter/mistralai/mistral-nemo:free","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-nemo":{"mode":"chat","base_model":"openrouter/mistralai/mistral-nemo","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4o-mini-2024-07-18":{"mode":"image_generation","base_model":"openrouter/openai/gpt-4o-mini-2024-07-18","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4o-mini":{"mode":"image_generation","base_model":"openrouter/openai/gpt-4o-mini","max_input_tokens":16384,"max_output_tokens":16384,"max_tokens":16384,"source":"merged_from_llm_models_csv","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":16384}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemma-2-27b-it":{"mode":"chat","base_model":"openrouter/google/gemma-2-27b-it","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemma-2-9b-it:free":{"mode":"chat","base_model":"openrouter/google/gemma-2-9b-it:free","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemma-2-9b-it":{"mode":"chat","base_model":"openrouter/google/gemma-2-9b-it","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3.5-sonnet-20240620:beta":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3.5-sonnet-20240620:beta","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3.5-sonnet-20240620":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3.5-sonnet-20240620","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/sao10k/l3-euryale-70b":{"mode":"chat","base_model":"openrouter/sao10k/l3-euryale-70b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/cognitivecomputations/dolphin-mixtral-8x22b":{"mode":"chat","base_model":"openrouter/cognitivecomputations/dolphin-mixtral-8x22b","provider":"litellm","max_input_tokens":16000,"max_output_tokens":16000,"max_tokens":16000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":16000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/qwen/qwen-2-72b-instruct":{"mode":"chat","base_model":"openrouter/qwen/qwen-2-72b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-7b-instruct-v0.3":{"mode":"chat","base_model":"openrouter/mistralai/mistral-7b-instruct-v0.3","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/nousresearch/hermes-2-pro-llama-3-8b":{"mode":"chat","base_model":"openrouter/nousresearch/hermes-2-pro-llama-3-8b","provider":"litellm","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":131072}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-7b-instruct:free":{"mode":"chat","base_model":"openrouter/mistralai/mistral-7b-instruct:free","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/microsoft/phi-3-mini-128k-instruct":{"mode":"chat","base_model":"openrouter/microsoft/phi-3-mini-128k-instruct","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/microsoft/phi-3-medium-128k-instruct":{"mode":"chat","base_model":"openrouter/microsoft/phi-3-medium-128k-instruct","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"openrouter/neversleep/llama-3-lumimaid-70b":{"mode":"chat","base_model":"openrouter/neversleep/llama-3-lumimaid-70b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/google/gemini-flash-1.5":{"mode":"image_generation","base_model":"openrouter/google/gemini-flash-1.5","provider":"litellm","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":131072,"range":{"min":1,"max":1000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-guard-2-8b":{"mode":"chat","base_model":"openrouter/meta-llama/llama-guard-2-8b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/gpt-4o:extended":{"mode":"image_generation","base_model":"openrouter/openai/gpt-4o:extended","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/sao10k/fimbulvetr-11b-v2":{"mode":"chat","base_model":"openrouter/sao10k/fimbulvetr-11b-v2","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/meta-llama/llama-3-8b-instruct":{"mode":"chat","base_model":"openrouter/meta-llama/llama-3-8b-instruct","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/microsoft/wizardlm-2-8x22b":{"mode":"chat","base_model":"openrouter/microsoft/wizardlm-2-8x22b","max_input_tokens":65536,"max_output_tokens":65536,"max_tokens":65536,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":32768,"range":{"min":1,"max":65536}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/google/gemini-pro-1.5":{"mode":"image_generation","base_model":"openrouter/google/gemini-pro-1.5","provider":"litellm","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":256,"range":{"min":1,"max":2000000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/openai/gpt-4-turbo":{"mode":"image_generation","base_model":"openrouter/openai/gpt-4-turbo","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/cohere/command-r-plus":{"mode":"chat","base_model":"openrouter/cohere/command-r-plus","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/cohere/command-r-plus-04-2024":{"mode":"chat","base_model":"openrouter/cohere/command-r-plus-04-2024","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/sophosympatheia/midnight-rose-70b":{"mode":"chat","base_model":"openrouter/sophosympatheia/midnight-rose-70b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/cohere/command":{"mode":"chat","base_model":"openrouter/cohere/command","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/cohere/command-r":{"mode":"chat","base_model":"openrouter/cohere/command-r","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3-haiku:beta":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3-haiku:beta","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3-opus:beta":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3-opus:beta","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/anthropic/claude-3-opus":{"mode":"image_generation","base_model":"openrouter/anthropic/claude-3-opus","provider":"litellm","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":200000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/cohere/command-r-03-2024":{"mode":"chat","base_model":"openrouter/cohere/command-r-03-2024","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/openai/gpt-3.5-turbo-0613":{"mode":"chat","base_model":"openrouter/openai/gpt-3.5-turbo-0613","max_input_tokens":4095,"max_output_tokens":4095,"max_tokens":4095,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4095}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/openai/gpt-4-turbo-preview":{"mode":"chat","base_model":"openrouter/openai/gpt-4-turbo-preview","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"merged_from_llm_models_csv","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"provider":"litellm","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/nousresearch/nous-hermes-2-mixtral-8x7b-dpo":{"mode":"chat","base_model":"openrouter/nousresearch/nous-hermes-2-mixtral-8x7b-dpo","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-small":{"mode":"chat","base_model":"openrouter/mistralai/mistral-small","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-tiny":{"mode":"chat","base_model":"openrouter/mistralai/mistral-tiny","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-7b-instruct-v0.2":{"mode":"chat","base_model":"openrouter/mistralai/mistral-7b-instruct-v0.2","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mixtral-8x7b-instruct":{"mode":"chat","base_model":"openrouter/mistralai/mixtral-8x7b-instruct","provider":"litellm","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":32768}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/neversleep/noromaid-20b":{"mode":"chat","base_model":"openrouter/neversleep/noromaid-20b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/alpindale/goliath-120b":{"mode":"chat","base_model":"openrouter/alpindale/goliath-120b","provider":"litellm","max_input_tokens":6144,"max_output_tokens":6144,"max_tokens":6144,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":6144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/undi95/toppy-m-7b":{"mode":"chat","base_model":"openrouter/undi95/toppy-m-7b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"openrouter/openrouter/auto":{"mode":"chat","base_model":"openrouter/openrouter/auto","max_input_tokens":2000000,"max_tokens":2000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_audio_input":true,"supports_video_input":true,"provider":"litellm","supports_pdf_input":true,"max_output_tokens":2000000,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":2000000}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/gpt-4-1106-preview":{"mode":"chat","base_model":"openrouter/openai/gpt-4-1106-preview","provider":"litellm","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":65536,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/mistralai/mistral-7b-instruct-v0.1":{"mode":"chat","base_model":"openrouter/mistralai/mistral-7b-instruct-v0.1","provider":"litellm","max_input_tokens":2824,"max_output_tokens":2824,"max_tokens":2824,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":1024,"range":{"min":1,"max":2824}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"openrouter/openai/gpt-3.5-turbo-instruct":{"mode":"chat","base_model":"openrouter/openai/gpt-3.5-turbo-instruct","max_input_tokens":4095,"max_output_tokens":4095,"max_tokens":4095,"source":"merged_from_llm_models_csv","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":false,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"litellm","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4095}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openrouter/pygmalionai/mythalion-13b":{"mode":"chat","base_model":"openrouter/pygmalionai/mythalion-13b","provider":"litellm","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":4096}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"source":"merged_from_llm_models_csv"},"openrouter/openai/gpt-4-0314":{"mode":"chat","base_model":"openrouter/openai/gpt-4-0314","provider":"litellm","max_input_tokens":8191,"max_output_tokens":8191,"max_tokens":8191,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8191}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cerebras/llama-4-scout-17b-16e-instruct":{"mode":"chat","base_model":"cerebras/llama-4-scout-17b-16e-instruct","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_function_calling":true,"supports_tool_choice":true,"source":"merged_from_llm_models_csv"},"cerebras/llama-4-maverick-17b-128e-instruct":{"mode":"chat","base_model":"cerebras/llama-4-maverick-17b-128e-instruct","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"seed","label":"Seed","helpText":"Seed value for reproducibility","type":"number","default":0,"range":{"min":0,"max":10000}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01}}],"source":"merged_from_llm_models_csv"},"cerebras/qwen-3-235b-a22b-thinking-2507":{"mode":"chat","base_model":"cerebras/qwen-3-235b-a22b-thinking-2507","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_reasoning":true,"source":"merged_from_llm_models_csv"},"cerebras/qwen-3-coder-480b":{"mode":"chat","base_model":"cerebras/qwen-3-coder-480b","provider":"litellm","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":2}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","default":50,"range":{"min":1,"max":100}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"repetition_penalty","label":"Repetition Penalty","helpText":"A number that controls the diversity of generated text by reducing the likelihood of repeated sequences. Higher values decrease repetition.","type":"number","default":1,"range":{"min":1,"max":2}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"source":"merged_from_llm_models_csv"},"vertex_ai/claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","deprecation_date":"2027-04-16","regional_endpoint_uplift_multiplier":1.1,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":2048,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","tool_use_system_prompt_tokens":346,"supports_minimal_reasoning_effort":true,"vertex_multi_region_only":true,"supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/claude-opus-4-7@default":{"mode":"chat","base_model":"claude-opus-4-7","deprecation_date":"2027-04-16","regional_endpoint_uplift_multiplier":1.1,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":2048,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","tool_use_system_prompt_tokens":346,"supports_minimal_reasoning_effort":true,"vertex_multi_region_only":true,"supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai/container-1g":{"mode":"chat","base_model":"container","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai/container-4g":{"mode":"chat","base_model":"container","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai/container-16g":{"mode":"chat","base_model":"container","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai/container-64g":{"mode":"chat","base_model":"container","provider":"openai","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_thinking_cache_preservation":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1,"fast":6},"supports_output_config":true,"supports_speed":false,"prompt_cache_min_tokens":2048,"source":"https://platform.claude.com/docs/en/about-claude/pricing","provider":"anthropic","supports_web_search":true,"deprecation_date":"2027-04-16","inference_geo_us_multiplier":1.1,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["auto","standard_only"],"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"tool_use_system_prompt_tokens":346},"claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"anthropic","deprecation_date":"2027-02-17","supports_web_search":true,"inference_geo_us_multiplier":1.1,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["auto","standard_only"],"supports_adaptive_thinking":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"anthropic.claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"}]},"us.anthropic.claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_adaptive_thinking":true,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"eu.anthropic.claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_adaptive_thinking":true,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"anthropic.claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":2048,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"tool_use_system_prompt_tokens":346,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.anthropic.claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":2048,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false,"tool_use_system_prompt_tokens":346},"eu.anthropic.claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":2048,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false,"tool_use_system_prompt_tokens":346},"gpt-5.4":{"mode":"chat","base_model":"gpt-5.4","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai","supports_reasoning_with_tool_calls":false,"supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_web_search":true,"supports_assistant_prefill":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.5":{"mode":"chat","base_model":"gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai","supports_reasoning_with_tool_calls":false,"supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true},"claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_thinking_cache_preservation":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1,"fast":6},"supports_output_config":true,"supports_speed":true,"supports_fast_mode":true,"prompt_cache_min_tokens":1024,"source":"https://platform.claude.com/docs/en/about-claude/pricing","provider":"anthropic","supports_web_search":true,"tool_use_system_prompt_tokens":346,"deprecation_date":"2027-05-28","inference_geo_us_multiplier":1.1,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["auto","standard_only"],"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":true,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true},"anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","tool_use_system_prompt_tokens":346,"supports_minimal_reasoning_effort":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"id":"stop_sequences"},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"us.anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","tool_use_system_prompt_tokens":346,"supports_minimal_reasoning_effort":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"anthropic/claude-opus-4.8":{"mode":"chat","base_model":"claude-opus-4-8","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"openrouter","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1},"supports_output_config":true,"prompt_cache_min_tokens":512,"supports_native_structured_output":true,"source":"https://docs.anthropic.com/en/docs/about-claude/models/overview","provider":"anthropic","tool_use_system_prompt_tokens":346,"supports_web_search":true,"deprecation_date":"2027-06-09","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning":{"can_disable":false},"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["auto","standard_only"],"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":true,"supports_native_effort":true,"supports_reasoning_disable":false,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"anthropic.claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","tool_use_system_prompt_tokens":346,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"reasoning":{"can_disable":false},"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"adaptive"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"gpt-5.4-2026-03-05":{"mode":"chat","base_model":"gpt-5.4","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai","supports_reasoning_with_tool_calls":false,"supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_web_search":true,"supports_assistant_prefill":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.4-mini":{"mode":"chat","base_model":"gpt-5.4-mini","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://developers.openai.com/api/docs/pricing","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","supports_reasoning_with_tool_calls":false,"supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.4-mini-2026-03-17":{"mode":"chat","base_model":"gpt-5.4-mini","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://developers.openai.com/api/docs/pricing","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","supports_reasoning_with_tool_calls":false,"supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.4-nano":{"mode":"chat","base_model":"gpt-5.4-nano","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://developers.openai.com/api/docs/pricing","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","supports_reasoning_with_tool_calls":false,"supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.4-nano-2026-03-17":{"mode":"chat","base_model":"gpt-5.4-nano","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://developers.openai.com/api/docs/pricing","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","supports_reasoning_with_tool_calls":false,"supports_service_tier":true,"model_parameters":[{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between -2.0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":-2,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"model-router":{"mode":"chat","base_model":"model-router","source":"https://azure.microsoft.com/en-us/pricing/details/ai-services/","comment":"Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)","provider":"azure","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"openai.gpt-5.4":{"mode":"responses","base_model":"openai.gpt-5.4","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_service_tier":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"openai.gpt-5.4-2026-03-05":{"mode":"responses","base_model":"openai.gpt-5.4","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_service_tier":true,"provider":"bedrock_mantle","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"openai.gpt-5.5":{"mode":"responses","base_model":"openai.gpt-5.5","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_web_search":true,"supports_reasoning_with_tool_calls":false},"openai.gpt-5.5-2026-04-23":{"mode":"responses","base_model":"openai.gpt-5.5","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_web_search":true,"supports_reasoning_with_tool_calls":false},"openai.gpt-5.6-luna":{"mode":"responses","base_model":"openai.gpt-5.6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":true,"provider":"bedrock_mantle","supports_reasoning_with_tool_calls":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"openai.gpt-5.6-terra":{"mode":"responses","base_model":"openai.gpt-5.6-terra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider":"bedrock_mantle","supports_reasoning_with_tool_calls":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"openai.gpt-5.6-sol":{"mode":"responses","base_model":"openai.gpt-5.6-sol","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider":"bedrock_mantle","supports_reasoning_with_tool_calls":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"glm-5p2":{"mode":"responses","base_model":"glm-5.2","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses","/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_web_search":true},"glm-5.2":{"mode":"responses","base_model":"glm-5.2","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses","/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":16384}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"supports_web_search":true},"gpt-5.6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"openai","supports_reasoning_with_tool_calls":false,"model":"gpt-5.6-sol","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true},"gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":true,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"openai","supports_reasoning_with_tool_calls":false,"model":"gpt-5.6-luna","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true},"gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"openai","supports_reasoning_with_tool_calls":false,"model":"gpt-5.6-terra","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true},"anthropic/claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1},"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"supports_output_config":true,"prompt_cache_min_tokens":1024,"provider":"anthropic","supports_web_search":true},"claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_thinking_cache_preservation":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1},"supports_output_config":true,"prompt_cache_min_tokens":1024,"source":"https://docs.anthropic.com/en/docs/about-claude/models/overview","provider":"anthropic","supports_web_search":true,"deprecation_date":"2027-06-30","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["auto","standard_only"],"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":true,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_thinking_cache_preservation":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1},"supports_output_config":true,"supports_speed":true,"supports_fast_mode":true,"prompt_cache_min_tokens":512,"source":"https://docs.anthropic.com/en/docs/about-claude/models/overview","provider":"anthropic","tool_use_system_prompt_tokens":286,"supports_web_search":true,"deprecation_date":"2027-07-24","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["auto","standard_only"],"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":true,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true},"anthropic/claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_speed":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1},"supports_output_config":true,"prompt_cache_min_tokens":512,"provider":"anthropic","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}],"tool_use_system_prompt_tokens":286,"supports_web_search":true},"gpt-5.3-codex":{"mode":"responses","base_model":"gpt-5.3-codex","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"openai","model_parameters":[{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":1},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","range":{"min":0,"max":1,"step":0.01},"default":1},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_service_tier":true,"conditionally_unsupported_fields":{"temperature":"when_effort_none","top_p":"when_effort_none"}},"gpt-5.5-2026-04-23":{"mode":"chat","base_model":"gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai","supports_reasoning_with_tool_calls":false,"supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_assistant_prefill":true},"bedrock_mantle/moonshotai.kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"supported_regions":["us-east-1","us-east-2","us-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-north-1","eu-west-2"],"max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_system_messages":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":8192}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"bedrock_mantle/moonshotai.kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-north-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4"],"max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.moonshot.ai/docs/guide/kimi-k2-5-quickstart","supports_reasoning":true,"supports_system_messages":true,"supports_video_input":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-3.1-flash-lite":{"mode":"chat","base_model":"gemini-3.1-flash-lite","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["minimal","low","medium","high"],"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini/gemini-3.1-flash-lite":{"mode":"chat","base_model":"gemini-3.1-flash-lite","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_audio_input":true,"supports_audio_output":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"gemini","source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"reasoning_effort_levels":["minimal","low","medium","high"],"supports_native_streaming":true,"supports_url_context":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/gemini-3.1-flash-lite":{"mode":"chat","base_model":"gemini-3.1-flash-lite","deprecation_date":"2027-05-07","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"service_tiers":["priority","flex"],"reasoning_effort_levels":["minimal","low","medium","high"],"supports_multimodal_tool_output":true,"supports_response_schema_with_tools":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-3.5-flash-lite":{"mode":"chat","base_model":"gemini-3.5-flash-lite","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["minimal","low","medium","high"],"supports_reasoning_disable":false,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":15,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_native_streaming":true,"tpm":250000,"web_search_billing_unit":"per_query","provider":"gemini"},"gemini/gemini-3.5-flash-lite":{"mode":"chat","base_model":"gemini-3.5-flash-lite","deprecation_date":"2027-07-21","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash-lite","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"gemini","rpm":15,"tpm":250000,"supports_code_execution":true,"supports_file_search":true,"supports_service_tier":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"vertex_ai/gemini-3.5-flash-lite":{"mode":"chat","base_model":"gemini-3.5-flash-lite","deprecation_date":"2027-07-21","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"regional_endpoint_uplift_multiplier":1.1,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","service_tiers":["priority","flex"],"supports_code_execution":true,"supports_file_search":true,"supports_service_tier":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":65535}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"gemini-3.1-pro-preview":{"mode":"chat","base_model":"gemini-3.1-pro-preview","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"media_resolution_medium","helpText":"Higher resolutions may provide better understanding but use more tokens.","id":"media_resolution","label":"Media Resolution","options":[{"label":"Low","value":"media_resolution_low"},{"label":"Medium","value":"media_resolution_medium"},{"label":"High","value":"media_resolution_high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high"],"supports_reasoning_disable":false,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini/gemini-3.1-pro-preview":{"mode":"chat","base_model":"gemini-3.1-pro","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.1-pro-preview","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_url_context":true,"supports_native_streaming":true,"tpm":800000,"web_search_billing_unit":"per_query","provider":"gemini","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"supports_audio_output":false,"supports_parallel_function_calling":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call. Supports Google Search, File Search, Code Execution, URL Context, and Function Calling.","type":"select"},{"id":"media_resolution","label":"Media Resolution","helpText":"Higher resolutions may provide better understanding but use more tokens.","type":"select","default":"media_resolution_medium","options":[{"label":"Low","value":"media_resolution_low"},{"label":"Medium","value":"media_resolution_medium"},{"label":"High","value":"media_resolution_high"}]},{"id":"thinking_level","label":"Thinking Level","helpText":"Set the thinking level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"temperature","label":"Temperature","helpText":"For Gemini 3, best results at default 1.0. Lower values may impact reasoning.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Output length","helpText":"Maximum number of tokens in response","type":"number","default":32768,"range":{"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":5,"minElements":1}},{"id":"topP","label":"Top P","helpText":"Nucleus sampling probability mass.","type":"number","range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"Text, JSON, or Structured output matching a supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema the model will follow.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Positive values penalize new tokens based on their existing frequency in the text so far.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"Number of most likely tokens to return at each position.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize tokens that appear in the text so far.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) best-effort deterministic sampling.","type":"number","range":{"min":0,"max":100}},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"stream","label":"Stream","helpText":"Whether the response is sent incrementally or as one complete result.","type":"boolean","default":false}]},"vertex_ai/gemini-3.1-pro-preview":{"mode":"chat","base_model":"gemini-3.1-pro","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_url_context":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","max_audio_length_hours":8.4,"max_audio_per_prompt":1,"max_images_per_prompt":3000,"max_pdf_size_mb":30,"max_video_length":1,"max_videos_per_prompt":10,"service_tiers":["priority","flex"],"supports_multimodal_tool_output":true,"supports_response_schema_with_tools":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call. Supports Google Search, File Search, Code Execution, URL Context, and Function Calling.","type":"select"},{"id":"media_resolution","label":"Media Resolution","helpText":"Higher resolutions may provide better understanding but use more tokens.","type":"select","default":"media_resolution_medium","options":[{"label":"Low","value":"media_resolution_low"},{"label":"Medium","value":"media_resolution_medium"},{"label":"High","value":"media_resolution_high"}]},{"id":"thinking_level","label":"Thinking Level","helpText":"Set the thinking level","type":"select","default":"high","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"temperature","label":"Temperature","helpText":"For Gemini 3, best results at default 1.0. Lower values may impact reasoning.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Output length","helpText":"Maximum number of tokens in response","type":"number","default":32768,"range":{"max":65536}},{"id":"stopSequences","label":"Stop Sequences","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":5,"minElements":1}},{"id":"topP","label":"Top P","helpText":"Nucleus sampling probability mass.","type":"number","range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","label":"Response Format","helpText":"Text, JSON, or Structured output matching a supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","label":"JSON Schema","helpText":"A JSON schema the model will follow.","type":"json","default":{"type":"object","properties":{},"required":[],"propertyOrdering":[]}}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}]},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Positive values penalize new tokens based on their existing frequency in the text so far.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"Number of most likely tokens to return at each position.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize tokens that appear in the text so far.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) best-effort deterministic sampling.","type":"number","range":{"min":0,"max":100}},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"stream","label":"Stream","helpText":"Whether the response is sent incrementally or as one complete result.","type":"boolean","default":false}]},"eu.anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"eu.anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"eu.anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"DeepSeek-V4-Flash-0731","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"DeepSeek-V4-Pro":{"mode":"chat","base_model":"DeepSeek-V4-Pro","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"DeepSeek-V4-Flash":{"mode":"chat","base_model":"DeepSeek-V4-Flash","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"DeepSeek-V3.2":{"mode":"chat","base_model":"DeepSeek-V3.2","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"DeepSeek-V3.2-Speciale":{"mode":"chat","base_model":"DeepSeek-V3.2-Speciale","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1},"supports_output_config":true,"prompt_cache_min_tokens":512,"supports_native_structured_output":true,"source":"https://platform.claude.com/docs/en/models/fable-5-1/whats-new-fable-5-1","provider":"anthropic","supports_forced_tool_choice":false,"supports_web_search":true,"inference_geo_us_multiplier":1.1,"tool_use_system_prompt_tokens":346,"deprecation_date":"2027-09-01","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning":{"can_disable":false},"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["auto","standard_only"],"supports_dynamic_reasoning_budget":false,"supports_mid_conversation_system_messages":true,"supports_native_effort":true,"supports_reasoning_disable":false,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"anthropic.claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_forced_tool_choice":false,"supports_web_search":true,"tool_use_system_prompt_tokens":346,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1","us-gov-east-1","us-gov-west-1"],"reasoning":{"can_disable":false},"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"adaptive"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"global.anthropic.claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1","us-gov-east-1","us-gov-west-1"],"supports_forced_tool_choice":false,"supports_web_search":true,"reasoning":{"can_disable":false},"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"us.anthropic.claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_forced_tool_choice":false,"supports_web_search":true,"tool_use_system_prompt_tokens":346,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1","us-gov-east-1","us-gov-west-1"],"reasoning":{"can_disable":false},"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"eu.anthropic.claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_forced_tool_choice":false,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"supports_native_structured_output":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"supports_web_search":true,"prompt_cache_min_tokens":512,"provider":"bedrock","reasoning":{"can_disable":false},"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"vertex_ai/claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"deprecation_date":"2027-03-01","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","supports_forced_tool_choice":false,"supports_web_search":true,"reasoning":{"can_disable":false},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-fable-5-1@default":{"mode":"chat","base_model":"claude-fable-5-1","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"deprecation_date":"2027-03-01","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","supports_forced_tool_choice":false,"supports_web_search":true,"reasoning":{"can_disable":false},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"deprecation_date":"2027-12-05","provider":"azure","supports_forced_tool_choice":false,"reasoning":{"can_disable":false},"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"global.anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"global.anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"anthropic.claude-mythos-preview":{"mode":"chat","base_model":"claude-mythos","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"thinking_always_on":true,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_output_config":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"au.anthropic.claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":2048,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"us.anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false,"supports_web_search":true},"us.anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false,"supports_web_search":true},"au.anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"au.anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false,"supports_web_search":true},"au.anthropic.claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_adaptive_thinking":true,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"au.anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false,"supports_web_search":true},"us.anthropic.claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"bedrock/us-gov-east-1/anthropic.claude-3-7-sonnet-20250219-v1:0":{"mode":"chat","base_model":"claude-3-7-sonnet","provider":"bedrock","supports_prompt_caching":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_assistant_prefill":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-east-1/anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-east-1/anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"bedrock_output_config_effort_ceiling":"xhigh","supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":1024,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"global.anthropic.claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"global.anthropic.claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":2048,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"global.anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"global.anthropic.claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_adaptive_thinking":true,"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"jp.anthropic.claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":2048,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"tool_use_system_prompt_tokens":346,"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false},"jp.anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_service_tier":true,"supports_speed":false,"supports_web_search":true},"jp.anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"jp.anthropic.claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","supports_adaptive_thinking":true,"supports_legacy_thinking":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_max_reasoning_effort":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"supports_web_search":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"reasoning_effort_levels":["low","medium","high","max"],"service_tiers":["default"],"supports_cache_point":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":true,"supports_speed":false},"jp.anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"us-gov.anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"bedrock_output_config_effort_ceiling":"xhigh","supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"bedrock/us-gov-east-1/anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-east-1/anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-east-1/anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"bedrock_output_config_effort_ceiling":"xhigh","supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":1024,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"us-gov.anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-east-1/us-gov.anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-west-1/anthropic.claude-haiku-4-5-20251001-v1:0":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/about-aws/whats-new/2025/10/claude-4-5-haiku-anthropic-amazon-bedrock","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-west-1/anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"bedrock_output_config_effort_ceiling":"xhigh","supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":1024,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"bedrock/us-gov-west-1/anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":64000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-west-1/anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":1024,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":8192,"range":{"min":1,"max":8192}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":0.7,"range":{"min":0,"max":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"top_k","label":"Top K","helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","type":"number","range":{"min":0,"max":100}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"bedrock/us-gov-west-1/anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","bedrock_converse_supports_strict_tools":false,"bedrock_output_config_effort_ceiling":"xhigh","supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":1024,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"bedrock/us-gov-west-1/us-gov.anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]}]},"eu.anthropic.claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_web_search":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"us-gov.anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","bedrock_converse_supports_strict_tools":false,"bedrock_output_config_effort_ceiling":"xhigh","supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":1024,"source":"https://aws.amazon.com/bedrock/pricing/","supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":16384,"range":{"min":1,"max":128000}},{"id":"stop_sequences","label":"Stop Sequence","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"output_format","label":"Output Format","helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","options":[{"label":"Output output","value":"json_schema","subFields":[{"id":"schema","label":"Schema","helpText":"A JSON schema that the model will follow when generating the response.","type":"json","default":{"type":"object"}}]}]},{"id":"metadata","label":"Metadata","helpText":"An external identifier for the user who is associated with the request.","type":"text"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"thinking","label":"Thinking","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}]},{"id":"output_config","label":"output_config","type":"select","accesorKey":"effort","default":{"type":"text"},"options":[{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}]}]},"azure/gpt-6-astra":{"mode":"chat","base_model":"gpt-6-astra","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"azure","supports_prompt_cache_breakpoints":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/gpt-6-astra-2026-09-03":{"mode":"chat","base_model":"gpt-6-astra","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://learn.microsoft.com/en-us/azure/ai-foundry/openai/concepts/models","supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"azure","supports_prompt_cache_breakpoints":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/us/gpt-6-astra":{"mode":"chat","base_model":"gpt-6-astra","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"azure","supports_prompt_cache_breakpoints":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"azure/us/gpt-6-astra-2026-09-03":{"mode":"chat","base_model":"gpt-6-astra","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"azure","supports_prompt_cache_breakpoints":true,"model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"response_format","type":"select","label":"Response Format","default":{"type":"text"},"options":[{"label":"Structured output","value":"json_schema","subFields":[{"id":"json_schema","type":"json","label":"JSON Schema","default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response."}]},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","accesorKey":"type"},{"id":"reasoning_effort","type":"select","label":"Reasoning Effort","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"helpText":"This parameter controls how many reasoning tokens the model generates before producing a response"},{"id":"verbosity","type":"select","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency."},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}]},"gpt-6-astra":{"mode":"chat","base_model":"gpt-6-astra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/models/gpt-6-astra","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"openai","supports_prompt_cache_breakpoints":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":false},"gemini-3.1-pro-preview-customtools":{"mode":"chat","base_model":"gemini-3.1-pro-preview-customtools","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high"],"supports_reasoning_disable":false,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini-3.1-flash-lite-preview":{"mode":"chat","base_model":"gemini-3.1-flash-lite-preview","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["minimal","low","medium","high"],"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini-3.5-flash":{"mode":"chat","base_model":"gemini-3.5-flash","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["minimal","low","medium","high"],"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_code_execution":true,"supports_file_search":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"gemini"},"gemini-3.6-flash":{"mode":"chat","base_model":"gemini-3.6-flash","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["minimal","low","medium","high"],"supports_reasoning_disable":false,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini-3.7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high"],"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini-3.8-flash":{"mode":"chat","base_model":"gemini-3.8-flash","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":false,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high"],"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"gemini-robotics-er-2-preview":{"mode":"chat","base_model":"gemini-robotics-er-2-preview","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":"high","helpText":"Set the thinking level","id":"thinking_level","label":"Thinking Level","options":[{"label":"Minimal","value":"minimal"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"min":1,"max":65536},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence. This technique helps to speed up the generation process and can improve the quality of the generated text by focusing on the most likely options.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"JSON","value":"json_object"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_web_search":true,"supports_service_tier":true,"service_tiers":["default","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["minimal","low","medium","high"],"supports_reasoning_disable":true,"supports_none_reasoning_effort":true,"supports_dynamic_reasoning_budget":true,"reasoning_budget":{"min":1,"max":65535},"supports_response_schema_with_tools":true},"chat-latest":{"mode":"chat","base_model":"chat","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"reasoning_effort_levels":["medium"],"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Medium","value":"medium"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"supports_service_tier":true,"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":false,"supports_reasoning_with_tool_calls":false},"codex-auto-review":{"mode":"chat","base_model":"gpt-5.4","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_service_tier":true,"supports_vision":true,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh"],"supports_web_search":true,"supports_assistant_prefill":true,"supports_none_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_reasoning_with_tool_calls":false},"gpt-5.4-pro":{"mode":"responses","base_model":"gpt-5.4-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["medium","high","xhigh"]},"gpt-5.4-pro-2026-03-05":{"mode":"responses","base_model":"gpt-5.4-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://developers.openai.com/api/docs/pricing","provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["medium","high","xhigh"]},"gpt-5.5-pro":{"mode":"responses","base_model":"gpt-5.5-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_low_reasoning_effort":false,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["medium","high","xhigh"]},"gpt-5.5-pro-2026-04-23":{"mode":"responses","base_model":"gpt-5.5-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_low_reasoning_effort":false,"provider":"openai","supports_service_tier":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["medium","high","xhigh"]},"gpt-5.6":{"mode":"chat","base_model":"gpt-5.6","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"openai","model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"None","value":"none"},{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_service_tier":true,"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":false},"claude-haiku-4-5@20251001":{"mode":"chat","base_model":"claude-haiku-4-5","deprecation_date":"2026-10-15","max_input_tokens":200000,"max_output_tokens":8192,"max_tokens":8192,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":8192,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":4096,"provider":"vertex_ai","regional_endpoint_uplift_multiplier":1.1,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude/haiku-4-5","supports_adaptive_thinking":false,"supports_assistant_prefill":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_native_streaming":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true},"claude-opus-4":{"mode":"chat","base_model":"claude-opus-4","deprecation_date":"2026-05-14","is_deprecated":true,"max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":32000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":1024,"provider":"vertex_ai","supports_adaptive_thinking":false,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tool_use_system_prompt_tokens":159},"claude-opus-4-1@20250805":{"mode":"chat","base_model":"claude-opus-4-1","deprecation_date":"2026-08-05","is_deprecated":true,"max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":32000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"provider":"vertex_ai","supports_adaptive_thinking":false,"supports_assistant_prefill":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true},"claude-opus-4-5@20251101":{"mode":"chat","base_model":"claude-opus-4-5","deprecation_date":"2026-11-24","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":4096,"provider":"vertex_ai-anthropic_models","reasoning_effort_levels":["low","medium","high"],"regional_endpoint_uplift_multiplier":1.1,"supports_adaptive_thinking":false,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_native_streaming":true,"supports_output_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tool_use_system_prompt_tokens":159},"claude-opus-4-6@default":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"provider":"vertex_ai","reasoning_effort_levels":["low","medium","high","max"],"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tool_use_system_prompt_tokens":346},"claude-opus-4-7@default":{"mode":"chat","base_model":"claude-opus-4-7","deprecation_date":"2027-04-16","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":2048,"provider":"vertex_ai","reasoning_effort_levels":["low","medium","high","xhigh","max"],"regional_endpoint_uplift_multiplier":1.1,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system_messages":false,"supports_minimal_reasoning_effort":true,"supports_native_effort":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":false,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"tool_use_system_prompt_tokens":346,"vertex_multi_region_only":true},"claude-opus-4-8@default":{"mode":"chat","base_model":"claude-opus-4-8","deprecation_date":"2027-05-28","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":1024,"provider":"vertex_ai","reasoning_effort_levels":["low","medium","high","xhigh","max"],"regional_endpoint_uplift_multiplier":1.1,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_mid_conversation_system_messages":true,"supports_native_effort":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":false,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true},"claude-opus-4@20250514":{"mode":"chat","base_model":"claude-opus-4","deprecation_date":"2026-05-14","is_deprecated":true,"max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":32000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":1024,"provider":"vertex_ai","supports_adaptive_thinking":false,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tool_use_system_prompt_tokens":159},"claude-sonnet-4":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-05-14","is_deprecated":true,"max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":1024,"provider":"vertex_ai","supports_adaptive_thinking":false,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tool_use_system_prompt_tokens":159},"claude-sonnet-4-5@20250929":{"mode":"chat","base_model":"claude-sonnet-4-5","deprecation_date":"2026-09-29","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":1024,"provider":"vertex_ai","regional_endpoint_uplift_multiplier":1.1,"supports_adaptive_thinking":false,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_native_streaming":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true},"claude-sonnet-4-6@default":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"provider":"vertex_ai","reasoning_effort_levels":["low","medium","high","max"],"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tool_use_system_prompt_tokens":346},"claude-sonnet-4@20250514":{"mode":"chat","base_model":"claude-sonnet-4","deprecation_date":"2026-05-14","is_deprecated":true,"max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 1. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":50,"helpText":"The top_k parameter is used to limit the number of choices for the next predicted word or token. It specifies the maximum number of tokens to consider at each step, based on their probability of occurrence.","id":"top_k","label":"Top K","range":{"max":100,"min":1},"type":"number"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"enabled","value":"enabled"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":1024,"provider":"vertex_ai","supports_adaptive_thinking":false,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_mid_conversation_system_messages":false,"supports_native_effort":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":true,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"tool_use_system_prompt_tokens":159},"claude-sonnet-5@default":{"mode":"chat","base_model":"claude-sonnet-5","deprecation_date":"2026-12-24","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}],"prompt_cache_min_tokens":1024,"provider":"vertex_ai","reasoning_effort_levels":["low","medium","high","xhigh","max"],"regional_endpoint_uplift_multiplier":1.1,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_dynamic_reasoning_budget":false,"supports_forced_tool_choice":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_mid_conversation_system_messages":true,"supports_native_effort":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_disable":true,"supports_reasoning_effort":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_response_schema_with_tools":true,"supports_sampling_params":false,"supports_service_tier":false,"supports_speed":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true},"bedrock_mantle/xai.grok-4.3":{"mode":"responses","base_model":"grok-4.3","use_openai_responses_path":true,"max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-west-2","us-east-1","us-east-2","us-gov-west-1"],"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"helpText":"Modify the likelihood of specified tokens appearing in the completion.","id":"logit_bias","label":"Logit Bias","range":{"max":100,"min":-100},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"A unique identifier representing your end-user.","id":"metadata","label":"Metadata","type":"text"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_reasoning_with_tool_calls":true,"supports_service_tier":true,"supports_web_search":false,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":false,"top_p":false,"web_search_options":false}},"bedrock_mantle/xai.grok-4.6":{"mode":"chat","base_model":"grok-4.6","use_openai_responses_path":true,"max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-west-2","us-gov-east-1","us-east-1","us-east-2","us-west-1","us-gov-west-1","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"helpText":"Modify the likelihood of specified tokens appearing in the completion.","id":"logit_bias","label":"Logit Bias","range":{"max":100,"min":-100},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"A unique identifier representing your end-user.","id":"metadata","label":"Metadata","type":"text"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_reasoning_with_tool_calls":true,"supports_service_tier":true,"supports_web_search":false,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":false,"top_p":false,"web_search_options":false}},"global.xai.grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"unsupported_fields":{"stop":true,"top_p":true}},"grok-3":{"mode":"chat","base_model":"grok-3","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-beta":{"mode":"chat","base_model":"grok-3","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-fast-beta":{"mode":"chat","base_model":"grok-3-fast","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-fast-latest":{"mode":"chat","base_model":"grok-3-fast","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-latest":{"mode":"chat","base_model":"grok-3","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-mini":{"mode":"chat","base_model":"grok-3-mini","deprecation_date":"2026-02-28","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-mini-beta":{"mode":"chat","base_model":"grok-3-mini","deprecation_date":"2026-02-28","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-mini-fast":{"mode":"chat","base_model":"grok-3-mini-fast","deprecation_date":"2026-02-28","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-mini-fast-beta":{"mode":"chat","base_model":"grok-3-mini-fast","deprecation_date":"2026-02-28","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-mini-fast-latest":{"mode":"chat","base_model":"grok-3-mini-fast","deprecation_date":"2026-02-28","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-3-mini-latest":{"mode":"chat","base_model":"grok-3-mini","deprecation_date":"2026-02-28","is_deprecated":true,"max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://x.ai/api#pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4":{"mode":"chat","base_model":"grok-4","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-0709":{"mode":"chat","base_model":"grok-4","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-1-fast":{"mode":"chat","base_model":"grok-4-1-fast","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","supports_assistant_prefill":true,"supports_audio_input":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-1-fast-non-reasoning":{"mode":"chat","base_model":"grok-4-1-fast-non","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning","supports_assistant_prefill":true,"supports_audio_input":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-1-fast-non-reasoning-latest":{"mode":"chat","base_model":"grok-4-1-fast-non","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models/grok-4-1-fast-non-reasoning","supports_assistant_prefill":true,"supports_audio_input":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-1-fast-reasoning":{"mode":"chat","base_model":"grok-4-1-fast","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","supports_assistant_prefill":true,"supports_audio_input":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-1-fast-reasoning-latest":{"mode":"chat","base_model":"grok-4-1-fast","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","supports_assistant_prefill":true,"supports_audio_input":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-fast-non-reasoning":{"mode":"chat","base_model":"grok-4-fast-non","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-fast-reasoning":{"mode":"chat","base_model":"grok-4-fast","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4-latest":{"mode":"chat","base_model":"grok-4","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-0309":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-0309-non-reasoning":{"mode":"chat","base_model":"grok-4.20-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-0309-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-beta":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-beta-0309":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-beta-0309-non-reasoning":{"mode":"chat","base_model":"grok-4.20-beta-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-beta-0309-reasoning":{"mode":"chat","base_model":"grok-4.20-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-beta-latest":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-beta-latest-non-reasoning":{"mode":"chat","base_model":"grok-4.20-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-beta-latest-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-beta-non-reasoning":{"mode":"chat","base_model":"grok-4.20-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-beta-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-experimental-beta-0304":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-experimental-beta-0304-non-reasoning":{"mode":"chat","base_model":"grok-4.20-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-experimental-beta-0304-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-experimental-beta-latest":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-experimental-beta-non-reasoning-latest":{"mode":"chat","base_model":"grok-4.20-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-experimental-beta-reasoning-latest":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-multi-agent":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"top_p":false}},"grok-4.20-multi-agent-0309":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"top_p":false}},"grok-4.20-multi-agent-beta-0309":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"top_p":false}},"grok-4.20-multi-agent-beta-latest":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"top_p":false}},"grok-4.20-multi-agent-experimental-beta-0304":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"top_p":false}},"grok-4.20-multi-agent-experimental-beta-latest":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"top_p":false}},"grok-4.20-multi-agent-latest":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"top_p":false}},"grok-4.20-non-reasoning":{"mode":"chat","base_model":"grok-4.20-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-non-reasoning-gv2":{"mode":"chat","base_model":"grok-4.20-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-non-reasoning-latest":{"mode":"chat","base_model":"grok-4.20-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"grok-4.20-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-reasoning-gv2":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.20-reasoning-latest":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.3":{"mode":"chat","base_model":"grok-4.3","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.3-latest":{"mode":"chat","base_model":"grok-4.3","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.5":{"mode":"chat","base_model":"grok-4.5","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","reasoning_effort_levels":["minimal","low","medium","high"],"source":"https://docs.x.ai/developers/models/grok-4.5","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.5-latest":{"mode":"chat","base_model":"grok-4.5","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","reasoning_effort_levels":["minimal","low","medium","high"],"source":"https://docs.x.ai/developers/models/grok-4.5","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","reasoning_effort_levels":["minimal","low","medium","high","xhigh"],"source":"https://docs.x.ai/developers/models/grok-4.6","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-build-0.1":{"mode":"chat","base_model":"grok-build-0.1","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/developers/models/grok-build-0.1","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-build-latest":{"mode":"chat","base_model":"grok-build","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","reasoning_effort_levels":["minimal","low","medium","high"],"source":"https://docs.x.ai/developers/models","supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-code-fast":{"mode":"chat","base_model":"grok-code-fast","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-code-fast-1":{"mode":"chat","base_model":"grok-code-fast-1","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"grok-code-fast-1-0825":{"mode":"chat","base_model":"grok-code-fast-1","deprecation_date":"2026-05-15","is_deprecated":true,"max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"xai","source":"https://docs.x.ai/docs/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"reasoning_effort":true,"stop":true,"top_p":false,"web_search_options":true}},"us.xai.grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":true,"unsupported_fields":{"stop":true,"top_p":true}},"vertex_ai/xai/grok-4.1-fast-non-reasoning":{"mode":"chat","base_model":"grok-4.1-fast-non-reasoning","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","supports_audio_input":true,"supports_reasoning":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":0,"helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","id":"frequency_penalty","label":"Frequency Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"default":0,"helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","id":"presence_penalty","label":"Presence Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_service_tier":false,"unsupported_fields":{"frequency_penalty":false,"presence_penalty":false,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"vertex_ai/xai/grok-4.1-fast-reasoning":{"mode":"chat","base_model":"grok-4.1-fast-non-reasoning","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models/grok-4-1-fast-reasoning","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","supports_audio_input":true,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":0,"helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","id":"frequency_penalty","label":"Frequency Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"default":0,"helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","id":"presence_penalty","label":"Presence Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_reasoning_with_tool_calls":true,"supports_service_tier":false,"unsupported_fields":{"frequency_penalty":false,"presence_penalty":false,"stop":false,"top_p":false,"web_search_options":true}},"vertex_ai/xai/grok-4.20-non-reasoning":{"mode":"chat","base_model":"grok-4.20-non","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"vertex_ai","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":0,"helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","id":"frequency_penalty","label":"Frequency Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"default":0,"helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","id":"presence_penalty","label":"Presence Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_reasoning":false,"supports_service_tier":false,"unsupported_fields":{"frequency_penalty":false,"presence_penalty":false,"reasoning_effort":true,"stop":false,"top_p":false,"web_search_options":true}},"vertex_ai/xai/grok-4.20-reasoning":{"mode":"chat","base_model":"grok-4.20-reasoning","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":0,"helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","id":"frequency_penalty","label":"Frequency Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"default":0,"helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","id":"presence_penalty","label":"Presence Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_reasoning_with_tool_calls":true,"supports_service_tier":false,"unsupported_fields":{"frequency_penalty":false,"presence_penalty":false,"stop":false,"top_p":false,"web_search_options":true}},"vertex_ai/xai/grok-4.3":{"mode":"chat","base_model":"grok-4.3","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://docs.x.ai/developers/models/grok-4.3","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","supports_web_search":false,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":0,"helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","id":"frequency_penalty","label":"Frequency Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"A unique identifier representing your end-user.","id":"metadata","label":"Metadata","type":"text"},{"default":0,"helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","id":"presence_penalty","label":"Presence Penalty","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_reasoning_with_tool_calls":true,"supports_service_tier":true,"unsupported_fields":{"frequency_penalty":false,"presence_penalty":false,"stop":false,"top_p":false,"web_search_options":true}},"vertex_ai/xai/grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":524288,"max_output_tokens":524288,"max_tokens":524288,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","id":"top_logprobs","label":"Top Logprobs","range":{"max":5,"min":0,"step":1},"type":"number"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"supports_assistant_prefill":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":true,"supports_reasoning_with_tool_calls":true,"supports_service_tier":true,"supports_web_search":false,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":true,"top_p":false,"web_search_options":true}},"xai.grok-4.3":{"mode":"responses","base_model":"grok-4.3","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"helpText":"Modify the likelihood of specified tokens appearing in the completion.","id":"logit_bias","label":"Logit Bias","range":{"max":100,"min":-100},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"A unique identifier representing your end-user.","id":"metadata","label":"Metadata","type":"text"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"bedrock_mantle","source":"https://aws.amazon.com/bedrock/pricing/","supported_endpoints":["/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":false,"top_p":false,"web_search_options":false},"use_openai_responses_path":true},"xai.grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"helpText":"Modify the likelihood of specified tokens appearing in the completion.","id":"logit_bias","label":"Logit Bias","range":{"max":100,"min":-100},"type":"number"},{"default":false,"helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","id":"logprobs","label":"Logprobs","type":"boolean"},{"helpText":"A unique identifier representing your end-user.","id":"metadata","label":"Metadata","type":"text"},{"helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","id":"seed","label":"Seed","range":{"max":100,"min":0},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop","label":"Stop","type":"array"},{"default":1,"helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","id":"temperature","label":"Temperature","range":{"max":2,"min":0,"step":0.01},"type":"number"},{"default":1,"helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","id":"top_p","label":"Top P","range":{"max":1,"min":0,"step":0.01},"type":"number"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":64000,"min":1},"type":"number"},{"default":65536,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_completion_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"default":8192,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_output_tokens","label":"Max Tokens","type":"number"},{"helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","id":"max_tool_calls","label":"Max Tool Calls","range":{"max":128,"min":1,"step":1},"type":"number"},{"helpText":"Static predicted output content, such as the contents of a text file being regenerated with minor changes.","id":"prediction","label":"Predicted Output","type":"text"},{"helpText":"Used to cache responses for similar requests to optimize your cache hit rates.","id":"prompt_cache_key","label":"Prompt Cache Key","type":"text"},{"helpText":"The retention policy for the prompt cache.","id":"prompt_cache_retention","label":"Prompt Cache Retention","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}],"type":"select"},{"helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","id":"verbosity","label":"Verbosity","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"default":"medium","helpText":"Select the level of reasoning effort. Higher levels may use more tokens and provide more thorough reasoning in the model's response.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}],"type":"select"},{"default":false,"helpText":"Whether to stream back partial progress.","id":"stream","label":"Stream","type":"boolean"}],"provider":"bedrock_mantle","supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_service_tier":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"unsupported_fields":{"frequency_penalty":true,"presence_penalty":true,"stop":false,"top_p":false,"web_search_options":false},"use_openai_responses_path":true},"amazon.titan-embed-g1-text-02":{"mode":"embedding","base_model":"titan-embed-g1-text-02","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":1536,"provider":"bedrock"},"azure/us-gov/text-embedding-3-large":{"mode":"embedding","base_model":"us-gov/text-embedding-3-large","max_input_tokens":8191,"max_tokens":8191,"provider":"azure"},"azure/us-gov/text-embedding-3-small":{"mode":"embedding","base_model":"us-gov/text-embedding-3-small","max_input_tokens":8191,"max_tokens":8191,"provider":"azure"},"bedrock/us-gov-west-1/amazon.nova-2-multimodal-embeddings-v1:0":{"mode":"embedding","base_model":"nova-2-multimodal-embeddings","max_input_tokens":8172,"max_tokens":8172,"output_vector_size":3072,"supports_embedding_image_input":true,"supports_image_input":true,"supports_video_input":true,"supports_audio_input":true,"provider":"bedrock"},"cloudflare/workers-ai/@cf/baai/bge-base-en-v1.5":{"mode":"embedding","base_model":"@cf/baai/bge-base-en-v1.5","max_input_tokens":153600,"max_tokens":153600,"provider":"cloudflare","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"cloudflare/workers-ai/@cf/baai/bge-large-en-v1.5":{"mode":"embedding","base_model":"@cf/baai/bge-large-en-v1.5","provider":"cloudflare","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"cloudflare/workers-ai/@cf/baai/bge-m3":{"mode":"embedding","base_model":"@cf/baai/bge-m3","max_input_tokens":60000,"max_tokens":60000,"provider":"cloudflare","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"cloudflare/workers-ai/@cf/baai/bge-small-en-v1.5":{"mode":"embedding","base_model":"@cf/baai/bge-small-en-v1.5","provider":"cloudflare","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"cloudflare/workers-ai/@cf/pfnet/plamo-embedding-1b":{"mode":"embedding","base_model":"@cf/pfnet/plamo-embedding-1b","provider":"cloudflare","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"cloudflare/workers-ai/@cf/qwen/qwen3-embedding-0.6b":{"mode":"embedding","base_model":"@cf/qwen/qwen3-embedding-0.6b","max_input_tokens":8192,"max_tokens":8192,"provider":"cloudflare","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"databricks/databricks-qwen3-embedding-0-6b":{"mode":"embedding","base_model":"qwen3-embedding-0.6b","max_input_tokens":32768,"max_tokens":32768,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Public Preview. Configurable output dimensionality up to 1024."},"output_vector_size":1024,"source":"https://www.databricks.com/product/pricing/foundation-model-serving","provider":"databricks"},"databricks/gte-large-en":{"mode":"embedding","base_model":"databricks-gte-large-en","provider":"databricks","source":"https://www.databricks.com/product/pricing/foundation-model-serving"},"databricks/qwen3-embedding-0-6b":{"mode":"embedding","base_model":"databricks-qwen3-embedding-0-6b","max_input_tokens":8192,"max_tokens":8192,"provider":"databricks","source":"https://www.databricks.com/product/pricing/foundation-model-serving"},"databricks/system.ai.gte-large-en":{"mode":"embedding","base_model":"databricks-gte-large-en","provider":"databricks","source":"https://www.databricks.com/product/pricing/foundation-model-serving"},"databricks/system.ai.qwen3-embedding-0-6b":{"mode":"embedding","base_model":"databricks-qwen3-embedding-0-6b","max_input_tokens":8192,"max_tokens":8192,"provider":"databricks","source":"https://www.databricks.com/product/pricing/foundation-model-serving"},"eu.twelvelabs.marengo-embed-3-0-v1:0":{"mode":"embedding","base_model":"marengo-embed-3-0","max_input_tokens":500,"max_tokens":500,"output_vector_size":512,"supports_embedding_image_input":true,"supports_image_input":true,"provider":"bedrock"},"gemini-embedding-2":{"mode":"embedding","base_model":"gemini-embedding-2","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":3072,"provider":"gemini","rpm":10000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supports_multimodal":true,"tpm":10000000},"gemini/gemini-embedding-2":{"mode":"embedding","base_model":"gemini-embedding-2","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":3072,"rpm":10000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supports_audio_input":true,"supports_multimodal":true,"supports_vision":true,"tpm":10000000,"provider":"gemini"},"gemini/gemini-embedding-2-preview":{"mode":"embedding","base_model":"gemini-embedding-2","provider":"gemini","max_input_tokens":8192,"max_tokens":8192,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_tool_choice":false,"supports_vision":true,"is_deprecated":true,"deprecation_date":"2026-08-10","output_vector_size":3072,"rpm":10000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supports_multimodal":true,"tpm":10000000},"gigachat/GigaEmbeddings-3B-2025-09":{"mode":"embedding","base_model":"gigaembeddings-3b-2025-09","max_input_tokens":4096,"max_tokens":4096,"output_vector_size":2048,"provider":"gigachat"},"global.cohere.embed-v4:0":{"mode":"embedding","base_model":"embed","max_input_tokens":128000,"max_tokens":128000,"output_vector_size":1536,"supports_embedding_image_input":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","sa-east-1"]},"libertai/bge-m3":{"mode":"embedding","base_model":"bge-m3","max_tokens":8192,"max_input_tokens":8192,"source":"https://docs.libertai.io/apis/text/","provider":"libertai"},"mistral/mistral-embed-2312":{"mode":"embedding","base_model":"mistral-embed","max_input_tokens":8192,"max_tokens":8192,"source":"https://docs.mistral.ai/models/mistral-embed-23-12","provider":"mistral"},"nebius/BAAI/bge-en-icl":{"mode":"embedding","base_model":"bge-en-icl","max_tokens":32768,"max_input_tokens":32768,"source":"https://nebius.com/prices","provider":"nebius"},"nebius/BAAI/bge-multilingual-gemma2":{"mode":"embedding","base_model":"bge-multilingual-gemma2","max_tokens":8192,"max_input_tokens":8192,"source":"https://nebius.com/prices","provider":"nebius"},"nebius/Qwen/Qwen3-Embedding-8B":{"mode":"embedding","base_model":"qwen3-embedding-8b","max_tokens":40960,"max_input_tokens":40960,"source":"https://tokenfactory.nebius.com/models/catalog/embedding/Qwen%2FQwen3-Embedding-8B","provider":"nebius"},"nebius/intfloat/e5-mistral-7b-instruct":{"mode":"embedding","base_model":"intfloat/e5-mistral-7b-instruct","max_tokens":32768,"max_input_tokens":32768,"source":"https://nebius.com/prices","provider":"nebius"},"oci/cohere.embed-english-image-v3.0":{"mode":"embedding","base_model":"embed-english-image","max_input_tokens":512,"max_tokens":512,"output_vector_size":1024,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_embedding_image_input":true,"provider":"oci"},"oci/cohere.embed-english-light-image-v3.0":{"mode":"embedding","base_model":"embed-english-light-image","max_input_tokens":512,"max_tokens":512,"output_vector_size":384,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_embedding_image_input":true,"provider":"oci"},"oci/cohere.embed-english-light-v3.0":{"mode":"embedding","base_model":"embed-english-light","max_input_tokens":512,"max_tokens":512,"output_vector_size":384,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","provider":"oci"},"oci/cohere.embed-english-v3.0":{"mode":"embedding","base_model":"embed-english","max_input_tokens":512,"max_tokens":512,"output_vector_size":1024,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","provider":"oci"},"oci/cohere.embed-multilingual-image-v3.0":{"mode":"embedding","base_model":"embed-multilingual-image","max_input_tokens":512,"output_vector_size":1024,"source":"https://www.oracle.com/artificial-intelligence/enterprise-ai/cost-estimator/","supports_vision":true,"provider":"oci"},"oci/cohere.embed-multilingual-light-image-v3.0":{"mode":"embedding","base_model":"embed-multilingual-light-image","max_input_tokens":512,"max_tokens":512,"output_vector_size":384,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_embedding_image_input":true,"provider":"oci"},"oci/cohere.embed-multilingual-light-v3.0":{"mode":"embedding","base_model":"embed-multilingual-light","max_input_tokens":512,"max_tokens":512,"output_vector_size":384,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","provider":"oci"},"oci/cohere.embed-multilingual-v3.0":{"mode":"embedding","base_model":"embed-multilingual","max_input_tokens":512,"max_tokens":512,"output_vector_size":1024,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","provider":"oci"},"oci/cohere.embed-v4.0":{"mode":"embedding","base_model":"embed","max_input_tokens":128000,"max_tokens":128000,"output_vector_size":1536,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_embedding_image_input":true,"provider":"oci"},"openrouter/baai/bge-base-en-v1.5":{"mode":"embedding","base_model":"bge-base-en-v1.5","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/baai/bge-large-en-v1.5":{"mode":"embedding","base_model":"bge-large-en-v1.5","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/baai/bge-m3":{"mode":"embedding","base_model":"bge-m3","provider":"openrouter","max_input_tokens":8194,"max_output_tokens":7374,"max_tokens":7374,"supports_response_schema":true},"openrouter/google/gemini-embedding-001":{"mode":"embedding","base_model":"gemini-embedding-001","provider":"openrouter","max_input_tokens":20000,"max_output_tokens":18000,"max_tokens":18000,"supports_response_schema":true},"openrouter/google/gemini-embedding-2":{"mode":"embedding","base_model":"gemini-embedding-2","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_vision":true,"supports_audio_input":true,"supports_pdf_input":true,"supports_response_schema":true},"openrouter/google/gemini-embedding-2-preview":{"mode":"embedding","base_model":"gemini-embedding-2-preview","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_vision":true,"supports_audio_input":true,"supports_pdf_input":true,"supports_response_schema":true},"openrouter/google/gemini-embedding-2:batch":{"mode":"embedding","base_model":"gemini-embedding-2:batch","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_vision":true,"supports_audio_input":true,"supports_pdf_input":true,"supports_response_schema":true},"openrouter/intfloat/e5-base-v2":{"mode":"embedding","base_model":"e5-base-v2","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/intfloat/e5-large-v2":{"mode":"embedding","base_model":"e5-large-v2","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/intfloat/multilingual-e5-large":{"mode":"embedding","base_model":"multilingual-e5-large","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/mistralai/codestral-embed-2505":{"mode":"embedding","base_model":"codestral-embed-2505","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":6553,"max_tokens":6553,"supports_response_schema":true},"openrouter/mistralai/mistral-embed-2312":{"mode":"embedding","base_model":"mistral-embed-2312","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":6553,"max_tokens":6553,"supports_response_schema":true},"openrouter/openai/text-embedding-3-large":{"mode":"embedding","base_model":"text-embedding-3-large","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_response_schema":true},"openrouter/openai/text-embedding-3-large:batch":{"mode":"embedding","base_model":"text-embedding-3-large:batch","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_response_schema":true},"openrouter/openai/text-embedding-3-small":{"mode":"embedding","base_model":"text-embedding-3-small","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_response_schema":true},"openrouter/openai/text-embedding-3-small:batch":{"mode":"embedding","base_model":"text-embedding-3-small:batch","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_response_schema":true},"openrouter/openai/text-embedding-ada-002":{"mode":"embedding","base_model":"text-embedding-ada-002","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_response_schema":true},"openrouter/openai/text-embedding-ada-002:batch":{"mode":"embedding","base_model":"text-embedding-ada-002:batch","provider":"openrouter","max_input_tokens":8192,"max_output_tokens":7372,"max_tokens":7372,"supports_response_schema":true},"openrouter/perplexity/pplx-embed-v1-0.6b":{"mode":"embedding","base_model":"pplx-embed-v1-0.6b","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800},"openrouter/perplexity/pplx-embed-v1-4b":{"mode":"embedding","base_model":"pplx-embed-v1-4b","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800},"openrouter/qwen/qwen3-embedding-4b":{"mode":"embedding","base_model":"qwen3-embedding-4b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":29491,"max_tokens":29491,"supports_response_schema":true},"openrouter/qwen/qwen3-embedding-8b":{"mode":"embedding","base_model":"qwen3-embedding-8b","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":29491,"max_tokens":29491,"supports_response_schema":true},"openrouter/sentence-transformers/all-minilm-l12-v2":{"mode":"embedding","base_model":"all-minilm-l12-v2","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/sentence-transformers/all-minilm-l6-v2":{"mode":"embedding","base_model":"all-minilm-l6-v2","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/sentence-transformers/all-mpnet-base-v2":{"mode":"embedding","base_model":"all-mpnet-base-v2","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/sentence-transformers/multi-qa-mpnet-base-dot-v1":{"mode":"embedding","base_model":"multi-qa-mpnet-base-dot-v1","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/sentence-transformers/paraphrase-minilm-l6-v2":{"mode":"embedding","base_model":"paraphrase-minilm-l6-v2","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/thenlper/gte-base":{"mode":"embedding","base_model":"gte-base","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/thenlper/gte-large":{"mode":"embedding","base_model":"gte-large","provider":"openrouter","max_input_tokens":512,"max_output_tokens":460,"max_tokens":460,"supports_response_schema":true},"openrouter/voyageai/voyage-4":{"mode":"embedding","base_model":"voyage-4","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800},"openrouter/voyageai/voyage-4-large":{"mode":"embedding","base_model":"voyage-4-large","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800},"openrouter/voyageai/voyage-4-lite":{"mode":"embedding","base_model":"voyage-4-lite","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800},"openrouter/voyageai/voyage-code-4":{"mode":"embedding","base_model":"voyage-code-4","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800},"openrouter/voyageai/voyage-multimodal-3.5":{"mode":"embedding","base_model":"voyage-multimodal-3.5","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800,"supports_vision":true},"perplexity/pplx-embed-context-v1-0.6b":{"mode":"embedding","base_model":"pplx-embed-context-v1-0.6b","max_input_tokens":32768,"max_tokens":32768,"output_vector_size":1024,"source":"https://docs.perplexity.ai/getting-started/pricing","provider":"perplexity"},"perplexity/pplx-embed-context-v1-4b":{"mode":"embedding","base_model":"pplx-embed-context-v1-4b","max_input_tokens":32768,"max_tokens":32768,"output_vector_size":2560,"source":"https://docs.perplexity.ai/getting-started/pricing","provider":"perplexity"},"perplexity/pplx-embed-v1-0.6b":{"mode":"embedding","base_model":"pplx-embed-v1-0.6b","max_input_tokens":32768,"max_tokens":32768,"output_vector_size":1024,"source":"https://docs.perplexity.ai/docs/embeddings/quickstart","provider":"perplexity"},"perplexity/pplx-embed-v1-4b":{"mode":"embedding","base_model":"pplx-embed-v1-4b","max_input_tokens":32768,"max_tokens":32768,"output_vector_size":2560,"source":"https://docs.perplexity.ai/docs/embeddings/quickstart","provider":"perplexity"},"scaleway/BAAI/bge-multilingual-gemma2":{"mode":"embedding","base_model":"bge-multilingual-gemma2","provider":"scaleway"},"scaleway/qwen/qwen3-embedding-8b":{"mode":"embedding","base_model":"qwen3-embedding-8b","provider":"scaleway"},"snowflake/snowflake-arctic-embed-l-v2.0":{"mode":"embedding","base_model":"snowflake-arctic-embed-l","max_tokens":8192,"max_input_tokens":8192,"provider":"snowflake"},"snowflake/snowflake-arctic-embed-m-v2.0":{"mode":"embedding","base_model":"snowflake-arctic-embed-m","max_tokens":8192,"max_input_tokens":8192,"provider":"snowflake"},"together_ai/intfloat/multilingual-e5-large-instruct":{"mode":"embedding","base_model":"intfloat/multilingual-e5-large-instruct","max_input_tokens":514,"max_tokens":514,"output_vector_size":1024,"provider":"together_ai","source":"https://docs.together.ai/docs/serverless-models"},"twelvelabs.marengo-embed-3-0-v1:0":{"mode":"embedding","base_model":"marengo-embed-3-0","max_input_tokens":500,"max_tokens":500,"output_vector_size":512,"supports_embedding_image_input":true,"supports_image_input":true,"provider":"bedrock"},"us.cohere.embed-v4:0":{"mode":"embedding","base_model":"embed","max_input_tokens":128000,"max_tokens":128000,"output_vector_size":1536,"supports_embedding_image_input":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","sa-east-1"]},"us.twelvelabs.marengo-embed-3-0-v1:0":{"mode":"embedding","base_model":"marengo-embed-3-0","max_input_tokens":500,"max_tokens":500,"output_vector_size":512,"supports_embedding_image_input":true,"supports_image_input":true,"provider":"bedrock"},"vertex_ai/gemini-embedding-2":{"mode":"embedding","base_model":"gemini-embedding-2","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":3072,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supports_multimodal":true,"uses_embed_content":true,"provider":"vertex_ai"},"vertex_ai/gemini-embedding-2-preview":{"mode":"embedding","base_model":"gemini-embedding-2","max_input_tokens":8192,"max_tokens":8192,"output_vector_size":3072,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supports_multimodal":true,"uses_embed_content":true,"provider":"vertex_ai"},"vertex_ai/gemini-flash-experimental":{"mode":"embedding","base_model":"gemini-flash","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"output_vector_size":3072,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","uses_embed_content":true,"provider":"vertex_ai"},"voyage/voyage-4":{"mode":"embedding","base_model":"voyage-4","max_input_tokens":32000,"max_tokens":32000,"output_vector_size":1024,"source":"https://docs.voyageai.com/docs/pricing","provider":"voyage"},"voyage/voyage-4-large":{"mode":"embedding","base_model":"voyage-4-large","max_input_tokens":32000,"max_tokens":32000,"output_vector_size":1024,"source":"https://docs.voyageai.com/docs/pricing","provider":"voyage"},"voyage/voyage-4-lite":{"mode":"embedding","base_model":"voyage-4-lite","max_input_tokens":32000,"max_tokens":32000,"output_vector_size":1024,"source":"https://docs.voyageai.com/docs/pricing","provider":"voyage"},"voyage/voyage-code-4":{"mode":"embedding","base_model":"voyage-code-4","max_input_tokens":32000,"max_tokens":32000,"output_vector_size":1024,"source":"https://docs.voyageai.com/docs/pricing","provider":"voyage"},"voyage/voyage-context-4":{"mode":"embedding","base_model":"voyage-context-4","max_input_tokens":120000,"max_tokens":120000,"output_vector_size":1024,"source":"https://docs.voyageai.com/docs/pricing","provider":"voyage"},"voyage/voyage-multilingual-2":{"mode":"embedding","base_model":"voyage-multilingual-2","max_input_tokens":32000,"max_tokens":32000,"output_vector_size":1024,"source":"https://docs.voyageai.com/docs/pricing","provider":"voyage"},"voyage/voyage-multimodal-3.5":{"mode":"embedding","base_model":"voyage-multimodal-3.5","max_input_tokens":32000,"max_tokens":32000,"output_vector_size":1024,"source":"https://docs.voyageai.com/docs/pricing","supports_embedding_image_input":true,"provider":"voyage"},"workers-ai/workers-ai/@cf/baai/bge-base-en-v1.5":{"mode":"embedding","base_model":"@cf/baai/bge-base-en-v1.5","max_input_tokens":153600,"max_tokens":153600,"provider":"workers-ai","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"workers-ai/workers-ai/@cf/baai/bge-large-en-v1.5":{"mode":"embedding","base_model":"@cf/baai/bge-large-en-v1.5","provider":"workers-ai","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"workers-ai/workers-ai/@cf/baai/bge-m3":{"mode":"embedding","base_model":"@cf/baai/bge-m3","max_input_tokens":60000,"max_tokens":60000,"provider":"workers-ai","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"workers-ai/workers-ai/@cf/baai/bge-small-en-v1.5":{"mode":"embedding","base_model":"@cf/baai/bge-small-en-v1.5","provider":"workers-ai","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"workers-ai/workers-ai/@cf/pfnet/plamo-embedding-1b":{"mode":"embedding","base_model":"@cf/pfnet/plamo-embedding-1b","provider":"workers-ai","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"workers-ai/workers-ai/@cf/qwen/qwen3-embedding-0.6b":{"mode":"embedding","base_model":"@cf/qwen/qwen3-embedding-0.6b","max_input_tokens":8192,"max_tokens":8192,"provider":"workers-ai","source":"https://developers.cloudflare.com/workers-ai/platform/pricing/"},"openai.gpt-6-astra":{"mode":"chat","base_model":"gpt-6-astra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":false,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock_mantle","reasoning_effort_levels":["low","medium","high","xhigh","max"],"source":"https://developers.openai.com/api/docs/models/gpt-6-astra","supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":false},"us.openai.gpt-6-astra":{"mode":"chat","base_model":"gpt-6-astra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_none_reasoning_effort":false,"supports_tool_choice":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"source":"https://developers.openai.com/api/docs/models/gpt-6-astra","provider":"bedrock_mantle","supports_prompt_cache_breakpoints":true,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"supports_computer_use":true,"supports_native_streaming":true,"supports_parallel_function_calling":false,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_response_schema":true,"supports_system_messages":true,"supports_web_search":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"supports_service_tier":true,"model_parameters":[{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","range":{"min":1,"max":10,"step":1},"default":1},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"max_completion_tokens","label":"Max Completion Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":128000}},{"id":"max_tool_calls","label":"Max Tool Calls","helpText":"The maximum number of total calls to built-in tools that can be processed in a response.","type":"number","range":{"min":1,"max":128,"step":1}},{"id":"prompt_cache_key","label":"Prompt Cache Key","helpText":"Used by OpenAI to cache responses for similar requests to optimize your cache hit rates.","type":"text"},{"id":"prompt_cache_retention","label":"Prompt Cache Retention","helpText":"The retention policy for the prompt cache.","type":"select","options":[{"label":"In Memory","value":"in_memory"},{"label":"24 Hours","value":"24h"}]},{"id":"verbosity","label":"Verbosity","helpText":"Verbosity determines how many output tokens are generated. Lowering the number of tokens reduces overall latency.","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"}]},{"id":"reasoning_effort","label":"Reasoning Effort","helpText":"This parameter controls how many reasoning tokens the model generates before producing a response","type":"select","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"}]},{"id":"reasoning_summary","label":"Reasoning Summary","helpText":"Controls the level of detail in the reasoning summary output","type":"select","options":[{"label":"Auto","value":"auto"},{"label":"Concise","value":"concise"},{"label":"Detailed","value":"detailed"}]},{"id":"response_format","label":"Response Format","helpText":"The format of the response. If set to 'json', the response will be a JSON object with the keys 'choices' and 'logprobs'. If set to 'text', the response will be a string with the completion text. If set to 'Structured output', the response will match the supplied JSON schema.","type":"select","accesorKey":"type","default":{"type":"text"},"options":[{"label":"Text","value":"text"},{"label":"JSON","value":"json_object"},{"label":"Structured output","value":"json_schema"}]},{"id":"web_search","label":"Web Search","helpText":"Enable web search tool for real-time information retrieval.","type":"boolean","default":false},{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false}],"service_tiers":["auto","flex","priority"],"supports_assistant_prefill":true,"supports_reasoning_with_tool_calls":false},"moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1000000,"max_tokens":1000000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"bedrock","model":"kimi-k3","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"min_output_tokens":16,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"us.moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1000000,"max_tokens":1000000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"bedrock","model":"kimi-k3","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"min_output_tokens":16,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"apac.moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1000000,"max_tokens":1000000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"bedrock","model":"kimi-k3","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"min_output_tokens":16,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"global.moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1000000,"max_tokens":1000000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"bedrock","model":"kimi-k3","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"min_output_tokens":16,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"eu.moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1000000,"max_tokens":1000000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"bedrock","model":"kimi-k3","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"min_output_tokens":16,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"promptTools","label":"Prompt Tools (Functions)","helpText":"A list of tools the model may call.","type":"select"},{"id":"temperature","label":"Temperature","helpText":"What sampling temperature to use, between 0 and 2. Higher values like 0.8 will make the output more random, while lower values like 0.2 will make it more focused and deterministic.","type":"number","default":1,"range":{"min":0,"max":2,"step":0.01}},{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":262144}},{"id":"stop","label":"Stop","helpText":"Custom text sequences that will cause the model to stop generating.","type":"array","array":{"type":"text","maxElements":4,"minElements":1}},{"id":"top_p","label":"Top P","helpText":"An alternative to sampling with temperature, called nucleus sampling, where the model considers the results of the tokens with top_p probability mass. So 0.1 means only the tokens comprising the top 10% probability mass are considered.","type":"number","default":1,"range":{"min":0,"max":1,"step":0.01}},{"id":"frequency_penalty","label":"Frequency Penalty","helpText":"Number between 0 and 2.0. Positive values penalize new tokens based on their existing frequency in the text so far, decreasing the model's likelihood to repeat the same line verbatim.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"logit_bias","label":"Logit Bias","helpText":"Modify the likelihood of specified tokens appearing in the completion.","type":"number","range":{"min":-100,"max":100}},{"id":"logprobs","label":"Logprobs","helpText":"Whether to return log probabilities of the output tokens or not. If true, returns the log probabilities of each output token returned in the content of message.","type":"boolean","default":false},{"id":"top_logprobs","label":"Top Logprobs","helpText":"An integer between 0 and 5 specifying the number of most likely tokens to return at each token position, each with an associated log probability.","type":"number","range":{"min":0,"max":5,"step":1}},{"id":"n","label":"Number of Responses","helpText":"How many chat completion choices to generate for each input message.","type":"number","default":1,"range":{"min":1,"max":10,"step":1}},{"id":"presence_penalty","label":"Presence Penalty","helpText":"Positive values penalize new tokens based on whether they appear in the text so far, increasing the model's likelihood to talk about new topics.","type":"number","range":{"min":0,"max":2,"step":0.01},"default":0},{"id":"seed","label":"Seed","helpText":"(BETA) - System will make a best effort to sample deterministically, such that repeated requests with the same seed and parameters should return the same result.","type":"number","range":{"min":0,"max":100}},{"id":"stream","label":"Stream","helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","type":"boolean","default":false},{"id":"metadata","label":"Metadata","helpText":"A unique identifier representing your end-user.","type":"text"}]},"glm-5p3":{"mode":"chat","base_model":"glm-5p3","supports_reasoning":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"reasoning_effort_renames":{"none":"low"},"supports_reasoning_disable":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"glm-5p3-flash":{"mode":"chat","base_model":"glm-5p3-flash","supports_reasoning":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"reasoning_effort_renames":{"none":"low"},"supports_reasoning_disable":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"routers/glm-5p3-fast":{"mode":"chat","base_model":"routers/glm-5p3-fast","supports_reasoning":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"reasoning_effort_renames":{"none":"low"},"supports_reasoning_disable":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"fireworks_ai/accounts/fireworks/models/glm-5p3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048576,"max_output_tokens":128000,"max_tokens":128000,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"reasoning_effort_renames":{"none":"low"},"supports_reasoning_disable":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"fireworks_ai/accounts/fireworks/routers/glm-5p3-fast":{"mode":"chat","base_model":"accounts/fireworks/routers/glm-5.3-fast","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"reasoning_effort_renames":{"none":"low"},"supports_reasoning_disable":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"fireworks_ai/glm-5p3-fast":{"mode":"chat","base_model":"glm-5.3-fast","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"reasoning_effort_renames":{"none":"low"},"supports_reasoning_disable":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"fireworks_ai/accounts/fireworks/models/glm-5p3-flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"supports_reasoning":true,"supports_reasoning_effort":true,"reasoning_effort_levels":["low","medium","high","xhigh","max"],"reasoning_effort_renames":{"none":"low"},"supports_reasoning_disable":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"fast":2},"supports_output_config":true,"supports_speed":true,"supports_fast_mode":true,"prompt_cache_min_tokens":512,"source":"https://platform.claude.com/docs/en/models/opus-5-5/overview","provider":"anthropic","inference_geo_us_multiplier":1.1,"default_reasoning_effort":"medium","supports_image_input":true,"supports_parallel_function_calling":true,"supports_forced_tool_choice":false,"supports_system_messages":true,"tool_use_system_prompt_tokens":286,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"reasoning_effort_levels":["low","medium","high","xhigh","max"],"supports_mid_conversation_system_messages":true,"supports_reasoning_disable":false,"model_parameters":[{"helpText":"A list of tools the model may call.","id":"promptTools","label":"Prompt Tools (Functions)","type":"select"},{"default":256,"helpText":"The maximum number of tokens that can be generated in the Result.","id":"max_tokens","label":"Max Tokens","range":{"max":128000,"min":1},"type":"number"},{"array":{"maxElements":4,"minElements":1,"type":"text"},"helpText":"Custom text sequences that will cause the model to stop generating.","id":"stop_sequences","label":"Stop Sequence","type":"array"},{"accesorKey":"type","default":{"type":"text"},"id":"thinking","label":"Thinking","options":[{"label":"disabled","value":"disabled"},{"label":"adaptive","value":"adaptive"}],"type":"select"},{"accesorKey":"effort","default":{"type":"text"},"id":"output_config","label":"output_config","options":[{"label":"max","value":"max"},{"label":"xhigh","value":"xhigh"},{"label":"high","value":"high"},{"label":"medium","value":"medium"},{"label":"low","value":"low"}],"type":"select"},{"default":"medium","helpText":"Reasoning effort gives the model guidance on how many reasoning tokens to generate before answering. Lower values favour speed and economical token usage; higher values favour more complete reasoning.","id":"reasoning_effort","label":"Reasoning Effort","options":[{"label":"Low","value":"low"},{"label":"Medium","value":"medium"},{"label":"High","value":"high"},{"label":"XHigh","value":"xhigh"},{"label":"Max","value":"max"}],"type":"select"},{"accesorKey":"type","default":{"type":"text"},"helpText":"The format of the response. If set to 'Structured output', the response will match the supplied JSON schema.","id":"response_format","label":"Response Format","options":[{"label":"Structured output","subFields":[{"default":{"name":"","schema":{},"strict":true},"helpText":"A JSON schema that the model will follow when generating the response.","id":"json_schema","label":"JSON Schema","type":"json"}],"value":"json_schema"},{"label":"Text","value":"text"}],"type":"select"},{"default":false,"helpText":"Enable web search tool for real-time information retrieval.","id":"web_search","label":"Web Search","type":"boolean"},{"default":false,"helpText":"The stream parameter in the API controls whether the response is sent in incremental updates, like tokenized data as it's generated, or as a complete result in one go.","id":"stream","label":"Stream","type":"boolean"}]},"writer.palmyra-vision-7b":{"mode":"chat","base_model":"palmyra-vision-7b","max_input_tokens":4096,"max_output_tokens":4096,"max_tokens":4096,"source":"https://aws.amazon.com/bedrock/pricing/","supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"global.twelvelabs.pegasus-1-2-v1:0":{"mode":"chat","base_model":"pegasus-1-2","supports_video_input":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"anthropic.claude-opus-5-5":{"mode":"chat","base_model":"anthropic.claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","thinking_always_on":true,"supports_forced_tool_use":false,"provider":"bedrock","default_reasoning_effort":"medium","supports_image_input":true,"supports_parallel_function_calling":true,"supports_forced_tool_choice":false,"supports_system_messages":true,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"global.anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","thinking_always_on":true,"supports_forced_tool_use":false,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us.anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","thinking_always_on":true,"supports_forced_tool_use":false,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"eu.anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"thinking_always_on":true,"supports_forced_tool_use":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"au.anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"thinking_always_on":true,"supports_forced_tool_use":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"jp.anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"thinking_always_on":true,"supports_forced_tool_use":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","deprecation_date":"2027-04-06","supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":2048,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","deprecation_date":"2027-12-05","supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","deprecation_date":"2027-07-08","supports_mid_conversation_system":true,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","deprecation_date":"2027-09-01","supports_mid_conversation_system":true,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":1024,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","deprecation_date":"2027-06-30","supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":1024,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"azure/gpt-5.5":{"mode":"chat","base_model":"gpt-5.5","deprecation_date":"2027-10-26","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-chat-latest":{"mode":"chat","base_model":"gpt-chat","deprecation_date":"2026-12-02","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"reasoning_effort_levels":["medium"],"source":"https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/","supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.5-2026-04-23":{"mode":"chat","base_model":"gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"deprecation_date":"2027-10-26","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.5-2026-04-24":{"mode":"chat","base_model":"gpt-5.5","deprecation_date":"2027-10-26","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","supports_computer_use":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.4":{"mode":"chat","base_model":"gpt-5.4","deprecation_date":"2027-09-02","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.4-2026-03-05":{"mode":"chat","base_model":"gpt-5.4","deprecation_date":"2027-09-02","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.4-pro":{"mode":"responses","base_model":"gpt-5.4-pro","deprecation_date":"2027-09-07","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.4-pro-2026-03-05":{"mode":"responses","base_model":"gpt-5.4-pro","deprecation_date":"2027-09-07","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.4-mini":{"mode":"chat","base_model":"gpt-5.4-mini","deprecation_date":"2027-09-21","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":false,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.4-mini-2026-03-17":{"mode":"chat","base_model":"gpt-5.4-mini","deprecation_date":"2027-09-21","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":false,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.4-nano":{"mode":"chat","base_model":"gpt-5.4-nano","deprecation_date":"2027-09-21","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":false,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.4-nano-2026-03-17":{"mode":"chat","base_model":"gpt-5.4-nano","deprecation_date":"2027-09-21","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":false,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/model-router":{"mode":"chat","base_model":"model-router","deprecation_date":"2027-05-20","max_input_tokens":200000,"max_output_tokens":32768,"max_tokens":32768,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-services/","comment":"Flat cost of $0.14 per M input tokens for Azure AI Foundry Model Router infrastructure. Use pattern: azure_ai/model_router/<deployment-name> where deployment-name is your Azure deployment (e.g., azure-model-router)","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"azure/gpt-audio-1.5-2026-02-23":{"mode":"chat","base_model":"gpt-audio-1.5","deprecation_date":"2027-08-24","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"azure/gpt-audio-mini":{"mode":"chat","base_model":"gpt-audio-mini","deprecation_date":"2027-04-06","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"azure/gpt-5.3-codex":{"mode":"responses","base_model":"gpt-5.3-codex","deprecation_date":"2027-08-24","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.4":{"mode":"chat","base_model":"gpt-5.4","deprecation_date":"2027-09-02","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.4":{"mode":"chat","base_model":"gpt-5.4","deprecation_date":"2027-09-02","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.4-2026-03-05":{"mode":"chat","base_model":"gpt-5.4","deprecation_date":"2027-09-02","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.4-2026-03-05":{"mode":"chat","base_model":"gpt-5.4","deprecation_date":"2027-09-02","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_none_reasoning_effort":true,"default_reasoning_effort":"none","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":true,"provider":"azure","supports_service_tier":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.6":{"mode":"chat","base_model":"gpt-5.6","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.6-sol-2026-07-09":{"mode":"chat","base_model":"gpt-5.6-sol","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.6-terra-2026-07-09":{"mode":"chat","base_model":"gpt-5.6-terra","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.6-luna-2026-07-09":{"mode":"chat","base_model":"gpt-5.6-luna","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/chat-latest":{"mode":"chat","base_model":"chat","deprecation_date":"2026-12-02","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"reasoning_effort_levels":["medium"],"source":"https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/","supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.6":{"mode":"chat","base_model":"gpt-5.6","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-chat-latest":{"mode":"chat","base_model":"gpt-chat","deprecation_date":"2026-12-02","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"reasoning_effort_levels":["medium"],"source":"https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/","supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.6":{"mode":"chat","base_model":"gpt-5.6","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","deprecation_date":"2028-01-11","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.5":{"mode":"chat","base_model":"gpt-5.5","deprecation_date":"2027-10-26","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.5":{"mode":"chat","base_model":"gpt-5.5","deprecation_date":"2027-10-26","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.5-2026-04-23":{"mode":"chat","base_model":"gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"deprecation_date":"2027-10-26","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us/gpt-5.5-2026-04-24":{"mode":"chat","base_model":"gpt-5.5","deprecation_date":"2027-10-26","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.5-2026-04-23":{"mode":"chat","base_model":"gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"deprecation_date":"2027-10-26","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/eu/gpt-5.5-2026-04-24":{"mode":"chat","base_model":"gpt-5.5","deprecation_date":"2027-10-26","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.5-pro":{"mode":"responses","base_model":"gpt-5.5-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_none_reasoning_effort":false,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_low_reasoning_effort":false,"provider":"azure","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/gpt-5.5-pro-2026-04-23":{"mode":"responses","base_model":"gpt-5.5-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/Codestral-2501":{"mode":"chat","base_model":"codestral","max_input_tokens":256000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_native_streaming":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"azure/FW-DeepSeek-V3.2":{"mode":"chat","base_model":"fw-deepseek-v3.2","deprecation_date":"2027-07-01","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"azure/FW-DeepSeek-V4-Pro":{"mode":"chat","base_model":"fw-deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"azure/FW-GLM-5":{"mode":"chat","base_model":"fw-glm-5","deprecation_date":"2027-07-01","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/FW-GLM-5.1":{"mode":"chat","base_model":"fw-glm-5.1","deprecation_date":"2027-07-01","max_input_tokens":202800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"azure/FW-GLM-5.2":{"mode":"chat","base_model":"fw-glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"azure/FW-GLM-5.2-Fast":{"mode":"chat","base_model":"fw-glm-5.2-fast","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"azure/FW-Inkling":{"mode":"chat","base_model":"fw-inkling","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"azure/FW-Kimi-K2.5":{"mode":"chat","base_model":"fw-kimi-k2.5","deprecation_date":"2027-07-01","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"azure/FW-Kimi-K2.6":{"mode":"chat","base_model":"fw-kimi-k2.6","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"azure/FW-Kimi-K2.7-Code":{"mode":"chat","base_model":"fw-kimi-k2.7-code","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"azure/FW-Kimi-K3":{"mode":"chat","base_model":"fw-kimi-k3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"azure/FW-MiniMax-M2.5":{"mode":"chat","base_model":"fw-minimax-m2.5","deprecation_date":"2027-07-01","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"azure/FW-MiniMax-M3":{"mode":"chat","base_model":"fw-minimax-m3","max_input_tokens":512000,"max_output_tokens":512000,"max_tokens":512000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":512000}}]},"azure/FW-Nemotron-Lightning-3.5-30B-A3B":{"mode":"chat","base_model":"fw-nemotron-lightning-3.5-30b-a3b","max_input_tokens":262144,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/FW-Nemotron-3-Ultra-NVFP4":{"mode":"chat","base_model":"fw-nemotron-3-ultra-nvfp4","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supported_endpoints":["/v1/chat/completions"],"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"azure/MAI-Thinking-1":{"mode":"chat","base_model":"mai-thinking-1","max_input_tokens":256000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"azure/cohere-command-a":{"mode":"chat","base_model":"command-a","max_input_tokens":131072,"max_output_tokens":8182,"max_tokens":8182,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_tool_choice":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8182}}]},"azure/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","deprecation_date":"2028-02-20","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_prompt_caching":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"azure/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","deprecation_date":"2028-02-20","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_prompt_caching":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"azure/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"DeepSeek-V4-Flash-0731","deprecation_date":"2026-12-03","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/deepseek/","supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"provider":"azure","supports_web_search":true,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supports_assistant_prefill":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_reasoning_with_tool_calls":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"azure/grok-4.3":{"mode":"chat","base_model":"grok-4.3","max_input_tokens":20000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"azure/grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/grok-4-20-reasoning":{"mode":"chat","base_model":"grok-4-20","deprecation_date":"2027-04-06","max_input_tokens":262000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_reasoning":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"azure/grok-4-20-non-reasoning":{"mode":"chat","base_model":"grok-4-20-non","deprecation_date":"2027-04-06","max_input_tokens":262000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"azure/grok-4-1-fast-non-reasoning":{"mode":"chat","base_model":"grok-4-1-fast-non","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"azure/grok-4-1-fast-reasoning":{"mode":"chat","base_model":"grok-4-1-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"azure/kimi-k2.6":{"mode":"chat","base_model":"kimi-k2.6","deprecation_date":"2027-04-16","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-kimi-k2-6-in-microsoft-foundry/4513125","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_prompt_caching":true,"provider":"azure","supports_video_input":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"bedrock/ap-northeast-1/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/ap-south-1/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/ap-southeast-2/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/ap-southeast-3/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/eu-north-1/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/eu-central-1/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/eu-west-1/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/eu-west-2/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/eu-west-2/nvidia.nemotron-super-3-120b":{"mode":"chat","base_model":"nvidia-nemotron-super-3-120b","max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock/eu-south-1/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/sa-east-1/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/us-east-1/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/us-east-2/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/us-gov-west-1/amazon.nova-lite-v1:0":{"mode":"chat","base_model":"nova-lite","max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":10000}}]},"bedrock/us-gov-west-1/amazon.nova-micro-v1:0":{"mode":"chat","base_model":"nova-micro","max_input_tokens":128000,"max_output_tokens":10000,"max_tokens":10000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":10000}}]},"bedrock/us-west-2/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"cerebras/qwen-3.8-27b":{"mode":"chat","base_model":"qwen3.8-27b","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://api.cerebras.ai/public/v1/models/qwen-3.8-27b","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"cerebras","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"claude-opus-4-6-20260205":{"mode":"chat","base_model":"claude-opus-4-6","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_legacy_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_thinking_cache_preservation":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_vision":true,"provider_specific_entry":{"us":1.1,"fast":6},"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_speed":true,"prompt_cache_min_tokens":4096,"provider":"anthropic","supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"claude-opus-4-7-20260416":{"mode":"chat","base_model":"claude-opus-4-7","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_thinking_cache_preservation":true,"supports_reasoning":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1,"fast":6},"supports_output_config":true,"supports_speed":true,"prompt_cache_min_tokens":2048,"provider":"anthropic","deprecation_date":"2027-04-16","supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/@cf/openai/gpt-oss-120b":{"mode":"chat","base_model":"","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"rpm":300,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/@cf/google/gemma-2b-it-lora":{"mode":"chat","base_model":"","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"cloudflare/@cf/meta/llama-3.2-3b-instruct":{"mode":"chat","base_model":"","max_input_tokens":80000,"max_output_tokens":80000,"max_tokens":80000,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":80000}}]},"cloudflare/@cf/meta/llama-guard-3-8b":{"mode":"chat","base_model":"","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"cloudflare/@cf/mistral/mistral-7b-instruct-v0.2-lora":{"mode":"chat","base_model":"","max_input_tokens":15000,"max_output_tokens":15000,"max_tokens":15000,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":15000}}]},"cloudflare/@cf/moonshotai/kimi-k2.7-code":{"mode":"chat","base_model":"","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"rpm":20,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"cloudflare/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b":{"mode":"chat","base_model":"","max_input_tokens":80000,"max_output_tokens":80000,"max_tokens":80000,"rpm":300,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":80000}}]},"cloudflare/@cf/meta/llama-3.1-8b-instruct-fp8":{"mode":"chat","base_model":"","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"cloudflare/@cf/meta/llama-3.2-1b-instruct":{"mode":"chat","base_model":"","max_input_tokens":60000,"max_output_tokens":60000,"max_tokens":60000,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":60000}}]},"cloudflare/@cf/moonshotai/kimi-k2.6":{"mode":"chat","base_model":"","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"rpm":20,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"cloudflare/@cf/zai-org/glm-4.7-flash":{"mode":"chat","base_model":"","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"rpm":300,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"cloudflare/@cf/meta-llama/llama-2-7b-chat-hf-lora":{"mode":"chat","base_model":"","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"cloudflare/@cf/meta/llama-3.3-70b-instruct-fp8-fast":{"mode":"chat","base_model":"","max_input_tokens":24000,"max_output_tokens":24000,"max_tokens":24000,"rpm":300,"supports_function_calling":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":24000}}]},"cloudflare/@cf/ibm-granite/granite-4.0-h-micro":{"mode":"chat","base_model":"","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"rpm":300,"supports_function_calling":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131000}}]},"cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct":{"mode":"chat","base_model":"","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"cloudflare/@cf/zai-org/glm-5.2":{"mode":"chat","base_model":"","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"rpm":20,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"cloudflare/@cf/nvidia/nemotron-3-120b-a12b":{"mode":"chat","base_model":"","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"rpm":300,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"mode":"chat","base_model":"","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/@cf/qwen/qwen3-30b-a3b-fp8":{"mode":"chat","base_model":"","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"rpm":300,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"cloudflare/@cf/google/gemma-7b-it-lora":{"mode":"chat","base_model":"","max_input_tokens":3500,"max_output_tokens":3500,"max_tokens":3500,"rpm":300,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":3500,"range":{"min":1,"max":3500}}]},"cloudflare/@cf/google/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"rpm":300,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"cloudflare/@cf/mistralai/mistral-small-3.1-24b-instruct":{"mode":"chat","base_model":"","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"rpm":300,"supports_function_calling":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/@cf/meta/llama-3.2-11b-vision-instruct":{"mode":"chat","base_model":"","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"rpm":300,"supports_vision":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/@cf/openai/gpt-oss-20b":{"mode":"chat","base_model":"","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"rpm":300,"supports_function_calling":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/@cf/meta/llama-4-scout-17b-16e-instruct":{"mode":"chat","base_model":"","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"rpm":300,"supports_function_calling":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131000}}]},"cloudflare/@cf/qwen/qwq-32b":{"mode":"chat","base_model":"","max_input_tokens":24000,"max_output_tokens":24000,"max_tokens":24000,"rpm":300,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":24000}}]},"command-a-plus-05-2026":{"mode":"chat","base_model":"command-a-plus","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://docs.cohere.com/docs/command-a-plus","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"cohere","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"dashscope/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"dashscope/deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"dashscope/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"dashscope/glm-5.1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":202745,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"dashscope/glm-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"dashscope/kimi-k2.7-code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":229376,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"dashscope/qwen3-max-2026-01-23":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"dashscope/qwen3-next-80b-a3b-instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"dashscope/qwen3-next-80b-a3b-thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"dashscope/qwen3-vl-235b-a22b-instruct":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"dashscope/qwen3-vl-235b-a22b-thinking":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"dashscope/qwen3-vl-32b-instruct":{"mode":"chat","base_model":"qwen3-vl-32b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"dashscope/qwen3-vl-32b-thinking":{"mode":"chat","base_model":"qwen3-vl-32b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"dashscope/qwen3-vl-plus":{"mode":"chat","base_model":"qwen3-vl-plus","max_input_tokens":260096,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":2e-7,"output_cost_per_token":0.0000016,"range":[0,32000]},{"input_cost_per_token":3e-7,"output_cost_per_token":0.0000024,"range":[32000,128000]},{"input_cost_per_token":6e-7,"output_cost_per_token":0.0000048,"range":[128000,256000]}],"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"dashscope/qwen3.5-plus":{"mode":"chat","base_model":"qwen3.5-plus","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_token":0.0000024,"range":[0,256000]},{"input_cost_per_token":5e-7,"output_cost_per_token":0.000003,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"dashscope/qwen3.7-max":{"mode":"chat","base_model":"qwen3.7-max","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"dashscope/qwen3.7-plus":{"mode":"chat","base_model":"qwen3.7-plus","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"cache_read_input_token_cost":8e-8,"input_cost_per_token":4e-7,"output_cost_per_token":0.0000016,"range":[0,256000]},{"cache_read_input_token_cost":2.4e-7,"input_cost_per_token":0.0000012,"output_cost_per_token":0.0000048,"range":[256000,1000000]}],"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"dashscope/qwen3.8-max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":991808,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"dashscope/qwen3.8-flash":{"mode":"chat","base_model":"qwen3.8-flash","max_input_tokens":991808,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.modelstudio.console.alibabacloud.com/en/model-studio/model-pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"dashscope/qwen3.8-omni-flash":{"mode":"chat","base_model":"qwen3.8-omni-flash","max_input_tokens":991808,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.modelstudio.console.alibabacloud.com/en/model-studio/model-pricing","supports_audio_input":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"provider":"dashscope","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwencloud/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"qwencloud/deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"qwencloud/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"qwencloud/glm-5.1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":202745,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwencloud/glm-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwencloud/kimi-k2.7-code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":229376,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen-coder":{"mode":"chat","base_model":"qwen-coder","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen-flash":{"mode":"chat","base_model":"qwen-flash","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"range":[0,256000]},{"input_cost_per_token":2.5e-7,"output_cost_per_token":0.000002,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen-flash-2025-07-28":{"mode":"chat","base_model":"qwen-flash","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"range":[0,256000]},{"input_cost_per_token":2.5e-7,"output_cost_per_token":0.000002,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen-max":{"mode":"chat","base_model":"qwen-max","max_input_tokens":30720,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"qwencloud/qwen-plus":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen-plus-2025-01-25":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"qwencloud/qwen-plus-2025-04-28":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen-plus-2025-07-14":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen-plus-2025-07-28":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen-plus-2025-09-11":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen-plus-latest":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen-turbo":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen-turbo-2024-11-01":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"qwencloud/qwen-turbo-2025-04-28":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen-turbo-latest":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen3-30b-a3b":{"mode":"chat","base_model":"qwen3-30b-a3b","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwencloud/qwen3-coder-flash":{"mode":"chat","base_model":"qwen3-coder-flash","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"cache_read_input_token_cost":8e-8,"input_cost_per_token":3e-7,"output_cost_per_token":0.0000015,"range":[0,32000]},{"cache_read_input_token_cost":1.2e-7,"input_cost_per_token":5e-7,"output_cost_per_token":0.0000025,"range":[32000,128000]},{"cache_read_input_token_cost":2e-7,"input_cost_per_token":8e-7,"output_cost_per_token":0.000004,"range":[128000,256000]},{"cache_read_input_token_cost":4e-7,"input_cost_per_token":0.0000016,"output_cost_per_token":0.0000096,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-coder-flash-2025-07-28":{"mode":"chat","base_model":"qwen3-coder-flash","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":3e-7,"output_cost_per_token":0.0000015,"range":[0,32000]},{"input_cost_per_token":5e-7,"output_cost_per_token":0.0000025,"range":[32000,128000]},{"input_cost_per_token":8e-7,"output_cost_per_token":0.000004,"range":[128000,256000]},{"input_cost_per_token":0.0000016,"output_cost_per_token":0.0000096,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-coder-plus":{"mode":"chat","base_model":"qwen3-coder-plus","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"cache_read_input_token_cost":1e-7,"input_cost_per_token":0.000001,"output_cost_per_token":0.000005,"range":[0,32000]},{"cache_read_input_token_cost":1.8e-7,"input_cost_per_token":0.0000018,"output_cost_per_token":0.000009,"range":[32000,128000]},{"cache_read_input_token_cost":3e-7,"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,256000]},{"cache_read_input_token_cost":6e-7,"input_cost_per_token":0.000006,"output_cost_per_token":0.00006,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-coder-plus-2025-07-22":{"mode":"chat","base_model":"qwen3-coder-plus","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.000001,"output_cost_per_token":0.000005,"range":[0,32000]},{"input_cost_per_token":0.0000018,"output_cost_per_token":0.000009,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,256000]},{"input_cost_per_token":0.000006,"output_cost_per_token":0.00006,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-max-preview":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-max":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-max-2026-01-23":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-next-80b-a3b-instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-next-80b-a3b-thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3-vl-235b-a22b-instruct":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen3-vl-235b-a22b-thinking":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen3-vl-32b-instruct":{"mode":"chat","base_model":"qwen3-vl-32b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen3-vl-32b-thinking":{"mode":"chat","base_model":"qwen3-vl-32b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen3-vl-plus":{"mode":"chat","base_model":"qwen3-vl-plus","max_input_tokens":260096,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":2e-7,"output_cost_per_token":0.0000016,"range":[0,32000]},{"input_cost_per_token":3e-7,"output_cost_per_token":0.0000024,"range":[32000,128000]},{"input_cost_per_token":6e-7,"output_cost_per_token":0.0000048,"range":[128000,256000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwencloud/qwen3.5-plus":{"mode":"chat","base_model":"qwen3.5-plus","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_token":0.0000024,"range":[0,256000]},{"input_cost_per_token":5e-7,"output_cost_per_token":0.000003,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3.7-max":{"mode":"chat","base_model":"qwen3.7-max","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3.7-plus":{"mode":"chat","base_model":"qwen3.7-plus","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"cache_read_input_token_cost":8e-8,"input_cost_per_token":4e-7,"output_cost_per_token":0.0000016,"range":[0,256000]},{"cache_read_input_token_cost":2.4e-7,"input_cost_per_token":0.0000012,"output_cost_per_token":0.0000048,"range":[256000,1000000]}],"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwencloud/qwen3.8-max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":991808,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwencloud/qwq-plus":{"mode":"chat","base_model":"qwq-plus","max_input_tokens":98304,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.qwencloud.com/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwencloud","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"qwen_ai_platform/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"qwen_ai_platform/deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"qwen_ai_platform/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"qwen_ai_platform/glm-5.1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":202745,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwen_ai_platform/glm-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwen_ai_platform/kimi-k2.7-code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":229376,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen-coder":{"mode":"chat","base_model":"qwen-coder","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen-flash":{"mode":"chat","base_model":"qwen-flash","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"range":[0,256000]},{"input_cost_per_token":2.5e-7,"output_cost_per_token":0.000002,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen-flash-2025-07-28":{"mode":"chat","base_model":"qwen-flash","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":5e-8,"output_cost_per_token":4e-7,"range":[0,256000]},{"input_cost_per_token":2.5e-7,"output_cost_per_token":0.000002,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen-max":{"mode":"chat","base_model":"qwen-max","max_input_tokens":30720,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"qwen_ai_platform/qwen-plus":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen-plus-2025-01-25":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"qwen_ai_platform/qwen-plus-2025-04-28":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen-plus-2025-07-14":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen-plus-2025-07-28":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen-plus-2025-09-11":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen-plus-latest":{"mode":"chat","base_model":"qwen-plus","max_input_tokens":997952,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_reasoning_token":0.000004,"output_cost_per_token":0.0000012,"range":[0,256000]},{"input_cost_per_token":0.0000012,"output_cost_per_reasoning_token":0.000012,"output_cost_per_token":0.0000036,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen-turbo":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen-turbo-2024-11-01":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"qwen_ai_platform/qwen-turbo-2025-04-28":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen-turbo-latest":{"mode":"chat","base_model":"qwen-turbo","max_input_tokens":1000000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen3-30b-a3b":{"mode":"chat","base_model":"qwen3-30b-a3b","max_input_tokens":129024,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"qwen_ai_platform/qwen3-coder-flash":{"mode":"chat","base_model":"qwen3-coder-flash","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"cache_read_input_token_cost":8e-8,"input_cost_per_token":3e-7,"output_cost_per_token":0.0000015,"range":[0,32000]},{"cache_read_input_token_cost":1.2e-7,"input_cost_per_token":5e-7,"output_cost_per_token":0.0000025,"range":[32000,128000]},{"cache_read_input_token_cost":2e-7,"input_cost_per_token":8e-7,"output_cost_per_token":0.000004,"range":[128000,256000]},{"cache_read_input_token_cost":4e-7,"input_cost_per_token":0.0000016,"output_cost_per_token":0.0000096,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-coder-flash-2025-07-28":{"mode":"chat","base_model":"qwen3-coder-flash","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":3e-7,"output_cost_per_token":0.0000015,"range":[0,32000]},{"input_cost_per_token":5e-7,"output_cost_per_token":0.0000025,"range":[32000,128000]},{"input_cost_per_token":8e-7,"output_cost_per_token":0.000004,"range":[128000,256000]},{"input_cost_per_token":0.0000016,"output_cost_per_token":0.0000096,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-coder-plus":{"mode":"chat","base_model":"qwen3-coder-plus","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"cache_read_input_token_cost":1e-7,"input_cost_per_token":0.000001,"output_cost_per_token":0.000005,"range":[0,32000]},{"cache_read_input_token_cost":1.8e-7,"input_cost_per_token":0.0000018,"output_cost_per_token":0.000009,"range":[32000,128000]},{"cache_read_input_token_cost":3e-7,"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,256000]},{"cache_read_input_token_cost":6e-7,"input_cost_per_token":0.000006,"output_cost_per_token":0.00006,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-coder-plus-2025-07-22":{"mode":"chat","base_model":"qwen3-coder-plus","max_input_tokens":997952,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.000001,"output_cost_per_token":0.000005,"range":[0,32000]},{"input_cost_per_token":0.0000018,"output_cost_per_token":0.000009,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,256000]},{"input_cost_per_token":0.000006,"output_cost_per_token":0.00006,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-max-preview":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-max":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-max-2026-01-23":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":258048,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"tiered_pricing":[{"input_cost_per_token":0.0000012,"output_cost_per_token":0.000006,"range":[0,32000]},{"input_cost_per_token":0.0000024,"output_cost_per_token":0.000012,"range":[32000,128000]},{"input_cost_per_token":0.000003,"output_cost_per_token":0.000015,"range":[128000,252000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-next-80b-a3b-instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-next-80b-a3b-thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3-vl-235b-a22b-instruct":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen3-vl-235b-a22b-thinking":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen3-vl-32b-instruct":{"mode":"chat","base_model":"qwen3-vl-32b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen3-vl-32b-thinking":{"mode":"chat","base_model":"qwen3-vl-32b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/model-pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen3-vl-plus":{"mode":"chat","base_model":"qwen3-vl-plus","max_input_tokens":260096,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":2e-7,"output_cost_per_token":0.0000016,"range":[0,32000]},{"input_cost_per_token":3e-7,"output_cost_per_token":0.0000024,"range":[32000,128000]},{"input_cost_per_token":6e-7,"output_cost_per_token":0.0000048,"range":[128000,256000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"qwen_ai_platform/qwen3.5-plus":{"mode":"chat","base_model":"qwen3.5-plus","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":4e-7,"output_cost_per_token":0.0000024,"range":[0,256000]},{"input_cost_per_token":5e-7,"output_cost_per_token":0.000003,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3.7-max":{"mode":"chat","base_model":"qwen3.7-max","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3.7-plus":{"mode":"chat","base_model":"qwen3.7-plus","max_input_tokens":991808,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tiered_pricing":[{"cache_read_input_token_cost":8e-8,"input_cost_per_token":4e-7,"output_cost_per_token":0.0000016,"range":[0,256000]},{"cache_read_input_token_cost":2.4e-7,"input_cost_per_token":0.0000012,"output_cost_per_token":0.0000048,"range":[256000,1000000]}],"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"qwen_ai_platform/qwen3.8-max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":991808,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwen_ai_platform/qwen3.8-flash":{"mode":"chat","base_model":"qwen3.8-flash","max_input_tokens":991808,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.modelstudio.console.alibabacloud.com/en/model-studio/model-pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwen_ai_platform/qwen3.8-omni-flash":{"mode":"chat","base_model":"qwen3.8-omni-flash","max_input_tokens":991808,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.modelstudio.console.alibabacloud.com/en/model-studio/model-pricing","supports_audio_input":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"qwen_ai_platform/qwq-plus":{"mode":"chat","base_model":"qwq-plus","max_input_tokens":98304,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.alibabacloud.com/help/en/model-studio/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"qwen_ai_platform","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"databricks/databricks-claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Text-only input. Prompts and responses retained 30 days for trust and safety. In-geo endpoint is 10% higher."},"prompt_cache_min_tokens":512,"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_adaptive_thinking":true,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_mid_conversation_system":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":false,"thinking_always_on":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"databricks/databricks-claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."},"prompt_cache_min_tokens":512,"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_mid_conversation_system":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"thinking_always_on":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-claude-opus-4-6":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_legacy_thinking":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"prompt_cache_min_tokens":4096,"provider":"databricks","supports_vision":true,"supports_minimal_reasoning_effort":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"databricks/databricks-claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"prompt_cache_min_tokens":2048,"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_adaptive_thinking":true,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_minimal_reasoning_effort":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"databricks/databricks-claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"prompt_cache_min_tokens":1024,"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_adaptive_thinking":true,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_mid_conversation_system":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_minimal_reasoning_effort":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Billed under the 'Claude Opus 4.5 / 4.6 / 4.7 / 4.8 / 5' row. In-geo endpoint is 10% higher."},"prompt_cache_min_tokens":512,"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_adaptive_thinking":true,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_mid_conversation_system":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"databricks/databricks-claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_legacy_thinking":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"prompt_cache_min_tokens":1024,"provider":"databricks","supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"databricks/databricks-claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. INTRODUCTORY PRICING through Aug 31, 2026. After that, Sonnet 5 reverts to the Sonnet 4.5/4.6 rates (42.857 input / 214.286 output DBU per 1M). Does not support temperature, top_p or top_k."},"prompt_cache_min_tokens":1024,"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_adaptive_thinking":true,"supports_assistant_prefill":true,"supports_function_calling":true,"supports_mid_conversation_system":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"databricks/databricks-deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Billing reads the per-token dollar fields; the '*_dbu_cost_per_token' fields are the published Databricks rates, kept for reference. Context/max output are the DeepSeek-published model limits (1M context, 384K max output)."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"databricks/databricks-deepseek-v4-pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Billing reads the per-token dollar fields; the '*_dbu_cost_per_token' fields are the published Databricks rates, kept for reference. Context/max output are the DeepSeek-published model limits (1M context, 384K max output)."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"databricks/databricks-gemini-3-1-flash-lite":{"mode":"chat","base_model":"gemini-3.1-flash-lite","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Rates exclude a 20% promotional discount in effect until Jan 31, 2027. In-geo endpoint is 10% higher."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_vision":true,"supports_audio_input":true,"supports_video_input":true,"provider":"databricks","supports_assistant_prefill":true,"supports_reasoning":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-gemini-3-1-flash-image":{"mode":"chat","base_model":"gemini-3.1-flash-image","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"metadata":{"notes":"Databricks DBU rates not yet published for this model; endpoint metadata only."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_vision":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"databricks/databricks-gemini-3-pro-image":{"mode":"chat","base_model":"gemini-3-pro-image","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"metadata":{"notes":"Databricks DBU rates not yet published for this model; endpoint metadata only."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_vision":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"databricks/databricks-gemini-3-1-pro":{"mode":"chat","base_model":"gemini-3.1-pro","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Billed under the 'Gemini 3.0 / 3.1 Pro' row. Long context (>200k tokens) tier: 71.429 input / 321.429 output DBU per 1M. Rates exclude a 20% promotional discount until Jan 31, 2027."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_reasoning":true,"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_vision":true,"supports_audio_input":true,"supports_video_input":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-gemini-3-flash":{"mode":"chat","base_model":"gemini-3-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Billed under the 'Gemini 3.0 Flash' row. Rates exclude a 20% promotional discount until Jan 31, 2027."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_vision":true,"supports_audio_input":true,"supports_video_input":true,"provider":"databricks","supports_assistant_prefill":true,"supports_reasoning":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-gemini-3-pro":{"mode":"chat","base_model":"gemini-3-pro","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-gemini-3-8-flash":{"mode":"chat","base_model":"gemini-3.8-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Databricks DBU rates not yet published for this model; endpoint metadata only."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-gemini-3-7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Databricks DBU rates not yet published for this model; endpoint metadata only."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-gemini-3-6-flash":{"mode":"chat","base_model":"gemini-3.6-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Rates exclude a 20% promotional discount in effect until Jan 31, 2027 (actual billed price is 20% lower)."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-gemini-3-5-flash":{"mode":"chat","base_model":"gemini-3.5-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Rates exclude a 20% promotional discount in effect until Jan 31, 2027. In-geo endpoint is 10% higher."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-gemini-3-5-flash-lite":{"mode":"chat","base_model":"gemini-3.5-flash-lite","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Rates exclude a 20% promotional discount in effect until Jan 31, 2027. In-geo endpoint is 10% higher."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-glm-5-2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Pay-per-token only; no provisioned throughput. 1M token context per Databricks docs."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"databricks/databricks-glm-5-3":{"mode":"chat","base_model":"glm-5-3","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"metadata":{"notes":"Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"thinking_always_on":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"databricks/databricks-glm-5-3-flash":{"mode":"chat","base_model":"glm-5-3-flash","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"metadata":{"notes":"Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"thinking_always_on":true,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"databricks/databricks-gpt-5-2":{"mode":"chat","base_model":"gpt-5.2","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"databricks","supports_assistant_prefill":true,"supports_reasoning":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-3-codex":{"mode":"chat","base_model":"gpt-5.3-codex","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Billed under the 'GPT 5.2/5.3 Codex' row. Responses API only; no batch inference."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"provider":"databricks","supports_assistant_prefill":true,"supports_reasoning":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-4":{"mode":"chat","base_model":"gpt-5.4","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Long context (>200k input) tier: 71.428 input / 321.429 output DBU per 1M."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-4-mini":{"mode":"chat","base_model":"gpt-5.4-mini","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-4-nano":{"mode":"chat","base_model":"gpt-5.4-nano","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Long context (>200k input) tier: 142.857 input / 642.857 output DBU per 1M. In-geo endpoint is 10% higher."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Long context (>200k input) tier: 57.143 input / 257.143 output DBU per 1M. In-geo endpoint is 10% higher."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Long context (>200k input) tier: 5.714 input / 25.714 output DBU per 1M. In-geo endpoint is 10% higher."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-5":{"mode":"chat","base_model":"gpt-5.5","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Long context (>200k input) tier: 142.857 input / 642.857 output DBU per 1M. Responses API only."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-gpt-5-5-pro":{"mode":"chat","base_model":"gpt-5.5-pro","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Billed under the 'GPT 5.4/5.5 Pro' row. No cache-read tier. Long context tier: 857.142 input / 3857.144 output DBU per 1M. Responses API only."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"databricks/databricks-grok-4-6":{"mode":"chat","base_model":"grok-4-6","max_input_tokens":500000,"metadata":{"notes":"Costs per token are the published Global DBU rates times $0.070 per DBU. The '*_dbu_cost_per_token' fields are provided for reference; cost calculation reads the dollar '*_cost_per_token' fields."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"databricks/databricks-inkling":{"mode":"chat","base_model":"inkling","max_input_tokens":1000000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Public Preview. Pay-per-token only. 1M token context per Databricks docs."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"thinking_always_on":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"databricks/databricks-kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1000000,"max_output_tokens":1048576,"max_tokens":1048576,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Pay-per-token only; no provisioned throughput. 1M token context per Databricks docs."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"databricks","supports_assistant_prefill":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"databricks/databricks-qwen35-122b-a10b":{"mode":"chat","base_model":"qwen3.5-122b-a10b","max_input_tokens":256000,"max_output_tokens":25000,"max_tokens":25000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Public Preview. Reasoning-only model; reasoning cannot be disabled. A 'Priority' tier is billed at 2x (6.286 input / 62.858 output DBU per 1M)."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"thinking_always_on":true,"provider":"databricks","supports_assistant_prefill":true,"supports_prompt_caching":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":25000}}]},"databricks/databricks-qwen3-next-80b-a3b-instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. Public Preview."},"source":"https://www.databricks.com/product/pricing/foundation-model-serving","supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":false,"provider":"databricks","max_input_tokens":262144,"supports_assistant_prefill":true,"supports_prompt_caching":false,"supports_reasoning":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"deepinfra/nvidia/NVIDIA-Nemotron-3.5-Lightning":{"mode":"chat","base_model":"nvidia-nemotron-3.5-lightning","max_input_tokens":262144,"source":"https://deepinfra.com/pricing","supports_tool_choice":true,"supports_function_calling":true,"supports_reasoning":true,"supports_prompt_caching":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"fal_ai/fal-ai/moondream3-preview/query":{"mode":"chat","base_model":"moondream3-preview/query","source":"https://fal.ai/models/fal-ai/moondream3-preview/query","supported_endpoints":["/v1/chat/completions"],"supports_reasoning":true,"supports_vision":true,"provider":"fal_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1048576,"max_output_tokens":384000,"max_tokens":384000,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro-0813","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/deepseek-v4-pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/models/glm-4p7":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":202800,"max_output_tokens":202800,"max_tokens":202800,"source":"https://fireworks.ai/models/fireworks/glm-4p7","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202800}}]},"fireworks_ai/accounts/fireworks/models/glm-5p1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":202800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/models/glm-5p2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/models/kimi-k2p6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/accounts/fireworks/models/kimi-k2p7-code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/accounts/fireworks/models/minimax-m2p1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":204800,"max_output_tokens":204800,"max_tokens":204800,"source":"https://fireworks.ai/models/fireworks/minimax-m2p1","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"fireworks_ai/accounts/fireworks/models/minimax-m3":{"mode":"chat","base_model":"minimax-m3","max_input_tokens":512000,"max_output_tokens":512000,"max_tokens":512000,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":512000}}]},"fireworks_ai/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1048576,"max_output_tokens":384000,"max_tokens":384000,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"fireworks_ai/glm-4p7":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":202800,"max_output_tokens":202800,"max_tokens":202800,"source":"https://fireworks.ai/models/fireworks/glm-4p7","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202800}}]},"fireworks_ai/glm-5p1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":202800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/glm-5p1-fast":{"mode":"chat","base_model":"glm-5.1-fast","max_input_tokens":202800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/glm-5p2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/kimi-k2p5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://fireworks.ai/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/kimi-k2p6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/kimi-k2p6-fast":{"mode":"chat","base_model":"kimi-k2.6-fast","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/kimi-k2p7-code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/kimi-k2p7-code-fast":{"mode":"chat","base_model":"kimi-k2.7-code-fast","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/minimax-m2p1":{"mode":"chat","base_model":"minimax-m2.1","max_input_tokens":204800,"max_output_tokens":204800,"max_tokens":204800,"source":"https://fireworks.ai/models/fireworks/minimax-m2p1","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"fireworks_ai/minimax-m3":{"mode":"chat","base_model":"minimax-m3","max_input_tokens":512000,"max_output_tokens":512000,"max_tokens":512000,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":512000}}]},"fireworks_ai/qwen3p7-plus":{"mode":"chat","base_model":"qwen3.7-plus","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"friendliai/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"supports_prompt_caching":true,"supports_reasoning":true,"reasoning_effort_levels":["low","high","max"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_image_input":true,"supports_video_input":true,"comment":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","source":"https://api.friendli.ai/serverless/v1/models","provider":"friendliai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"friendliai/zai-org/GLM-5.3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"supports_prompt_caching":true,"supports_reasoning":true,"reasoning_effort_levels":["low","high","max"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"supports_image_input":false,"supports_video_input":false,"comment":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","source":"https://api.friendli.ai/serverless/v1/models","provider":"friendliai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"friendliai/google/gemma-4-31B-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_prompt_caching":false,"supports_reasoning":true,"reasoning_effort_levels":[],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_image_input":true,"supports_video_input":false,"comment":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","source":"https://api.friendli.ai/serverless/v1/models","provider":"friendliai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"friendliai/zai-org/GLM-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"supports_prompt_caching":true,"supports_reasoning":true,"reasoning_effort_levels":["high","max"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"supports_image_input":false,"supports_video_input":false,"comment":"Open flagship GLM for long-horizon coding agents and million-token context work","source":"https://api.friendli.ai/serverless/v1/models","provider":"friendliai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"friendliai/deepseek-ai/DeepSeek-V3.2":{"mode":"chat","base_model":"deepseek-v3.2","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_prompt_caching":true,"supports_reasoning":true,"reasoning_effort_levels":[],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"supports_image_input":false,"supports_video_input":false,"comment":"DeepSeek chat model for instruction following, coding, and analysis","source":"https://api.friendli.ai/serverless/v1/models","provider":"friendliai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"friendliai/MiniMaxAI/MiniMax-M2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":196608,"max_output_tokens":196608,"max_tokens":196608,"supports_prompt_caching":true,"supports_reasoning":true,"reasoning_effort_levels":[],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"supports_image_input":false,"supports_video_input":false,"comment":"Prior MiniMax coding model for agent workflows, office edits, and automation","source":"https://api.friendli.ai/serverless/v1/models","provider":"friendliai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":196608}}]},"friendliai/zai-org/GLM-5.1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":202752,"max_output_tokens":202752,"max_tokens":202752,"supports_prompt_caching":true,"supports_reasoning":true,"reasoning_effort_levels":[],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"supports_image_input":false,"supports_video_input":false,"comment":"Strong GLM coding model for agentic engineering, terminals, and repository generation","source":"https://api.friendli.ai/serverless/v1/models","provider":"friendliai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"vertex_ai/gemini-3.5-flash":{"mode":"chat","base_model":"gemini-3.5-flash","prompt_cache_min_tokens":4096,"deprecation_date":"2027-05-19","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"regional_endpoint_uplift_multiplier":1.1,"source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","supports_audio_output":false,"supports_code_execution":true,"supports_file_search":true,"supports_service_tier":true,"service_tiers":["priority","flex"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"vertex_ai/gemini-3.6-flash":{"mode":"chat","base_model":"gemini-3.6-flash","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"regional_endpoint_uplift_multiplier":1.1,"source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.6-flash","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","supports_audio_output":false,"supports_code_execution":true,"supports_file_search":true,"supports_service_tier":true,"service_tiers":["priority","flex"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"vertex_ai/gemini-3.7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"regional_endpoint_uplift_multiplier":1.1,"source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.7-flash","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_minimal_reasoning_effort":false,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","supports_audio_output":false,"supports_code_execution":true,"supports_file_search":true,"supports_service_tier":true,"supports_computer_use":true,"service_tiers":["priority","flex"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"vertex_ai/gemini-3.8-flash":{"mode":"chat","base_model":"gemini-3.8-flash","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"regional_endpoint_uplift_multiplier":1.1,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_minimal_reasoning_effort":false,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","service_tiers":["priority","flex"],"supports_code_execution":true,"supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"vertex_ai/gemini-3.1-pro-preview-customtools":{"mode":"chat","base_model":"gemini-3.1-pro-customtools","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_url_context":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"gemini/gemini-robotics-er-2-preview":{"mode":"chat","base_model":"gemini-robotics-er-2","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"gemini/gemini-3-pro-image":{"mode":"image_generation","base_model":"gemini-3-pro-image","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"rpm":1000,"tpm":4000000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","supports_reasoning":false,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"gemini/nano-banana-pro-preview":{"mode":"image_generation","base_model":"nano-banana-pro","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"rpm":1000,"tpm":4000000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"gemini/gemini-3.1-flash-image":{"mode":"image_generation","base_model":"gemini-3.1-flash-image","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"rpm":1000,"tpm":4000000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"gemini/gemini-3.1-flash-lite-image":{"mode":"image_generation","base_model":"gemini-3.1-flash-lite-image","max_input_tokens":65536,"max_output_tokens":66000,"max_tokens":66000,"rpm":null,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-3.1-flash-lite-image","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":false,"tpm":null,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":66000}}]},"gemini/gemini-3.5-flash":{"mode":"chat","base_model":"gemini-3.5-flash","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.5-flash","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"tpm":800000,"web_search_billing_unit":"per_query","provider":"gemini","supports_code_execution":true,"supports_file_search":true,"supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"gemini/gemini-3.6-flash":{"mode":"chat","base_model":"gemini-3.6-flash","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"tpm":800000,"web_search_billing_unit":"per_query","provider":"gemini","supports_code_execution":true,"supports_file_search":true,"supports_service_tier":true,"supports_computer_use":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"gemini/gemini-3.7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/models/gemini-3.7-flash?s","supported_endpoints":["/v1/chat/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_minimal_reasoning_effort":false,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"tpm":800000,"web_search_billing_unit":"per_query","provider":"gemini","supports_code_execution":true,"supports_computer_use":true,"supports_file_search":true,"supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"gemini/gemini-3.8-flash":{"mode":"chat","base_model":"gemini-3.8-flash","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_output":false,"supports_audio_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_minimal_reasoning_effort":false,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"tpm":800000,"web_search_billing_unit":"per_query","provider":"gemini","service_tiers":["priority","flex"],"supports_code_execution":true,"supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"gemini/gemini-3.1-pro-preview-customtools":{"mode":"chat","base_model":"gemini-3.1-pro-customtools","prompt_cache_min_tokens":4096,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_url_context":true,"supports_native_streaming":true,"tpm":800000,"web_search_billing_unit":"per_query","provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"gemini-omni-flash-preview":{"mode":"chat","base_model":"gemini-omni-flash","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","video"],"supports_audio_input":true,"supports_reasoning":true,"supports_system_messages":true,"supports_video_input":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65535}}]},"gemini/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"gemma-4-26b-a4b-it","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://ai.google.dev/gemini-api/docs/pricing","provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"gemini/gemma-4-31b-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://ai.google.dev/gemini-api/docs/pricing","provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"gemini/lyria-3-clip-preview":{"mode":"chat","base_model":"lyria-3-clip","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_input":false,"supports_audio_output":true,"supports_function_calling":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_vision":false,"supports_web_search":false,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"gemini/lyria-3-pro-preview":{"mode":"chat","base_model":"lyria-3-pro","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_input":false,"supports_audio_output":true,"supports_function_calling":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_vision":false,"supports_web_search":false,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"github_copilot/claude-opus-4.6-fast":{"mode":"chat","base_model":"claude-opus-4-6-fast","supports_adaptive_thinking":true,"supports_legacy_thinking":true,"max_input_tokens":128000,"max_output_tokens":16000,"max_tokens":16000,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16000}}]},"github_copilot/gpt-5.3-codex":{"mode":"responses","base_model":"gpt-5.3-codex","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"github_copilot","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"github_copilot/mai-code-1-flash":{"mode":"chat","base_model":"mai-code-1-flash","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"provider":"github_copilot","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"github_copilot/mai-code-1-flash-internal":{"mode":"chat","base_model":"mai-code-1-flash-internal","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"provider":"github_copilot","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"chatgpt/gpt-5.5":{"mode":"responses","base_model":"gpt-5.5","source":"https://platform.openai.com/docs/models/gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"chatgpt/gpt-5.6-luna":{"mode":"responses","base_model":"gpt-5.6-luna","source":"https://platform.openai.com/docs/models/gpt-5.6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"chatgpt/gpt-5.6-sol":{"mode":"responses","base_model":"gpt-5.6-sol","source":"https://platform.openai.com/docs/models/gpt-5.6-sol","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"chatgpt/gpt-5.6-terra":{"mode":"responses","base_model":"gpt-5.6-terra","source":"https://platform.openai.com/docs/models/gpt-5.6-terra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"chatgpt/gpt-5.4":{"mode":"responses","base_model":"gpt-5.4","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"chatgpt/gpt-5.4-pro":{"mode":"responses","base_model":"gpt-5.4-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"chatgpt/gpt-5.3-codex":{"mode":"responses","base_model":"gpt-5.3-codex","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"chatgpt/gpt-5.3-codex-spark":{"mode":"responses","base_model":"gpt-5.3-codex-spark","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"chatgpt/gpt-5.3-instant":{"mode":"responses","base_model":"gpt-5.3-instant","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"chatgpt/gpt-5.3-chat-latest":{"mode":"responses","base_model":"gpt-5.3-chat","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_response_schema":true,"supports_vision":true,"provider":"chatgpt","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"gigachat/GigaChat-2":{"mode":"chat","base_model":"gigachat-2","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_system_messages":true,"provider":"gigachat","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"baseten/MiniMaxAI/MiniMax-M2.5":{"mode":"chat","base_model":"minimax-m2.5","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"baseten/nvidia/Nemotron-120B-A12B":{"mode":"chat","base_model":"nemotron-120b-a12b","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"baseten/zai-org/GLM-5":{"mode":"chat","base_model":"glm-5","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"baseten/zai-org/GLM-4.7":{"mode":"chat","base_model":"glm-4.7","max_input_tokens":200000,"max_output_tokens":200000,"max_tokens":200000,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":200000}}]},"baseten/zai-org/GLM-4.6":{"mode":"chat","base_model":"glm-4.6","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"baseten/moonshotai/Kimi-K2.5":{"mode":"chat","base_model":"kimi-k2.5","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"baseten/moonshotai/Kimi-K2-Thinking":{"mode":"chat","base_model":"kimi-k2-thinking","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"baseten/moonshotai/Kimi-K2-Instruct-0905":{"mode":"chat","base_model":"kimi-k2-instruct","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"baseten/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128072,"max_output_tokens":128072,"max_tokens":128072,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128072}}]},"baseten/deepseek-ai/DeepSeek-V3.1":{"mode":"chat","base_model":"deepseek-v3.1","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"baseten/deepseek-ai/DeepSeek-V3-0324":{"mode":"chat","base_model":"deepseek-v3","provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"gpt-audio-1.5":{"mode":"chat","base_model":"gpt-audio-1.5","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"openai","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"gpt-image-2.5-flare":{"mode":"image_generation","base_model":"gpt-image-2.5-flare","supported_endpoints":["/v1/images/generations","/v1/images/edits","/v1/responses"],"supports_vision":true,"supports_pdf_input":true,"source":"https://developers.openai.com/api/docs/models/gpt-image-2.5-flare","provider":"openai","supported_modalities":["text","image"],"supported_output_modalities":["image"],"supports_image_input":true,"supports_prompt_caching":true,"supports_function_calling":false,"supports_tool_choice":false,"supports_response_schema":false,"supports_native_streaming":false,"supports_reasoning":false,"metadata":{"notes":"Fastest GPT Image 2.5 variant for everyday generation. Quality settings: low, medium, high, xhigh, max, auto. Image cached input is $2.00/1M (0.000002 per token); text output is not billed. Token rates match gpt-image-2. No batch pricing published for this model. Rate limits are tier-based (Tier 1: 100k TPM / 5 IPM, Tier 5: 8M TPM / 250 IPM); Free tier not supported. Released 2026-09-08."},"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"gpt-image-2.5-flare-2026-09-08":{"mode":"image_generation","base_model":"gpt-image-2.5-flare","supported_endpoints":["/v1/images/generations","/v1/images/edits","/v1/responses"],"supports_vision":true,"supports_pdf_input":true,"source":"https://developers.openai.com/api/docs/models/gpt-image-2.5-flare","provider":"openai","supported_modalities":["text","image"],"supported_output_modalities":["image"],"supports_image_input":true,"supports_prompt_caching":true,"supports_function_calling":false,"supports_tool_choice":false,"supports_response_schema":false,"supports_native_streaming":false,"supports_reasoning":false,"metadata":{"notes":"Dated snapshot of gpt-image-2.5-flare. Same pricing and capabilities as the undated alias. Image cached input is $2.00/1M (0.000002 per token)."},"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"gpt-image-2.5-sunburst":{"mode":"image_generation","base_model":"gpt-image-2.5-sunburst","supported_endpoints":["/v1/images/generations","/v1/images/edits","/v1/responses"],"supports_vision":true,"supports_pdf_input":true,"source":"https://developers.openai.com/api/docs/models/gpt-image-2.5-sunburst","provider":"openai","supported_modalities":["text","image"],"supported_output_modalities":["image"],"supports_image_input":true,"supports_prompt_caching":true,"supports_function_calling":false,"supports_tool_choice":false,"supports_response_schema":false,"supports_native_streaming":false,"supports_reasoning":false,"metadata":{"notes":"Most capable GPT Image 2.5 variant (editing precision). Quality settings: low, medium, high, xhigh, max, auto. Image cached input is $2.00/1M (0.000002 per token); text output is not billed. Token rates match gpt-image-2. No batch pricing published for this model. Rate limits are tier-based (Tier 1: 100k TPM / 5 IPM, Tier 5: 8M TPM / 250 IPM); Free tier not supported. Released 2026-09-08."},"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"gpt-image-2.5-sunburst-2026-09-08":{"mode":"image_generation","base_model":"gpt-image-2.5-sunburst","supported_endpoints":["/v1/images/generations","/v1/images/edits","/v1/responses"],"supports_vision":true,"supports_pdf_input":true,"source":"https://developers.openai.com/api/docs/models/gpt-image-2.5-sunburst","provider":"openai","supported_modalities":["text","image"],"supported_output_modalities":["image"],"supports_image_input":true,"supports_prompt_caching":true,"supports_function_calling":false,"supports_tool_choice":false,"supports_response_schema":false,"supports_native_streaming":false,"supports_reasoning":false,"metadata":{"notes":"Dated snapshot of gpt-image-2.5-sunburst. Same pricing and capabilities as the undated alias. Image cached input is $2.00/1M (0.000002 per token)."},"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"gpt-6-sol":{"mode":"chat","base_model":"gpt-6-sol","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/models/gpt-6-sol","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"openai","regional_endpoint_uplift_multiplier":1.1,"default_reasoning_effort":"medium","reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_image_input":true,"supports_reasoning_with_tool_calls":true,"supports_low_reasoning_effort":true,"supports_native_structured_output":true,"supports_file_search":true,"supports_tool_search":true,"supports_service_tier":true,"metadata":{"notes":"For prompts over 272K input tokens, OpenAI charges 2x input/cache and 1.5x output for the full request. Batch and Flex are 50% of Standard. Fast is 2x the applicable rate. The current Bifrost schema has no dedicated Fast-above-272K pricing fields, so the Fast fields here represent short-context Fast pricing."},"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"gpt-6-luna":{"mode":"chat","base_model":"gpt-6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/models/gpt-6-luna","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_cache_breakpoint":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"openai","regional_endpoint_uplift_multiplier":1.1,"default_reasoning_effort":"medium","reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_image_input":true,"supports_reasoning_with_tool_calls":true,"supports_low_reasoning_effort":true,"supports_native_structured_output":true,"supports_file_search":true,"supports_tool_search":true,"supports_service_tier":true,"metadata":{"notes":"For prompts over 272K input tokens, OpenAI charges 2x input/cache and 1.5x output for the full request. Batch and Flex are 50% of Standard. Fast is 2x the applicable rate. The current Bifrost schema has no dedicated Fast-above-272K pricing fields, so the Fast fields here represent short-context Fast pricing."},"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"gpt-5.6-cyber":{"mode":"chat","base_model":"gpt-5.6-cyber","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"source":"https://developers.openai.com/api/docs/pricing","supports_computer_use":true,"supports_parallel_function_calling":true,"provider":"openai","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"daybreak-red-latest":{"mode":"chat","base_model":"daybreak-red","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"source":"https://developers.openai.com/api/docs/models/gpt-daybreak-red-latest","supports_computer_use":true,"supports_parallel_function_calling":true,"provider":"openai","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"gpt-daybreak-red-latest":{"mode":"responses","base_model":"gpt-daybreak-red","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_native_streaming":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"source":"https://developers.openai.com/api/docs/models/gpt-daybreak-red-latest","supports_computer_use":true,"supports_parallel_function_calling":true,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"daybreak-blue-latest":{"mode":"chat","base_model":"daybreak-blue","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"source":"https://developers.openai.com/api/docs/models/gpt-daybreak-blue-latest","supports_parallel_function_calling":true,"provider":"openai","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"gpt-daybreak-blue-latest":{"mode":"responses","base_model":"gpt-daybreak-blue","max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_computer_use":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"source":"https://developers.openai.com/api/docs/models/gpt-daybreak-blue-latest","supports_parallel_function_calling":true,"provider":"openai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"groq/llama-guard-3-8b":{"mode":"chat","base_model":"llama-guard-3-8b","max_input_tokens":8192,"max_tokens":8192,"source":"https://console.groq.com/docs/model/llama-guard-3-8b","provider":"groq","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"groq/meta-llama/llama-prompt-guard-2-22m":{"mode":"chat","base_model":"llama-prompt-guard-2-22m","max_input_tokens":512,"max_output_tokens":512,"max_tokens":512,"source":"https://console.groq.com/docs/models","provider":"groq","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":512}}]},"groq/meta-llama/llama-prompt-guard-2-86m":{"mode":"chat","base_model":"llama-prompt-guard-2-86m","max_input_tokens":512,"max_output_tokens":512,"max_tokens":512,"source":"https://console.groq.com/docs/model/meta-llama/llama-prompt-guard-2-86m","provider":"groq","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":512,"range":{"min":1,"max":512}}]},"groq/openai/gpt-oss-safeguard-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_web_search":true,"provider":"groq","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"crusoe/deepseek-ai/DeepSeek-R1-0528":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":false,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":false,"provider":"crusoe","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"crusoe/deepseek-ai/DeepSeek-V3-0324":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":163840,"max_output_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"crusoe","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"crusoe/google/gemma-3-12b-it":{"mode":"chat","base_model":"gemma-3-12b-it","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"crusoe","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"crusoe/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"crusoe","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"crusoe/moonshotai/Kimi-K2-Thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":false,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":false,"provider":"crusoe","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"crusoe/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"crusoe","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"crusoe/Qwen/Qwen3-235B-A22B-Instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"crusoe","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"inception/mercury-2":{"mode":"chat","base_model":"mercury-2","max_input_tokens":128000,"max_output_tokens":50000,"max_tokens":50000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"inception","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":50000}}]},"inception/mercury-2.5":{"mode":"chat","base_model":"mercury-2.5","max_input_tokens":260000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://docs.inceptionlabs.ai/get-started/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"inception","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"text-completion-inception/mercury-edit-2":{"mode":"completion","base_model":"mercury-edit-2","max_input_tokens":32000,"max_output_tokens":8192,"max_tokens":8192,"provider":"text-completion-inception","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"meta/muse-spark-1.1":{"mode":"chat","base_model":"muse-spark-1.1","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://ai.developer.meta.com/docs/pricing-rate-limits","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/messages"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"meta","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"meta/muse-spark-1.2":{"mode":"chat","base_model":"muse-spark-1.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://ai.developer.meta.com/docs/pricing-rate-limits","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/messages"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"meta","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"meta/muse-spark-1.2-contributor":{"mode":"chat","base_model":"muse-spark-1.2-contributor","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://ai.developer.meta.com/docs/pricing-rate-limits","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/messages"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"meta","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"meta/muse-spark-1.3":{"mode":"chat","base_model":"muse-spark-1.3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://ai.developer.meta.com/docs/pricing-rate-limits","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/messages"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"meta","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"meta/muse-spark-1.3-contributor":{"mode":"chat","base_model":"muse-spark-1.3-contributor","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://ai.developer.meta.com/docs/pricing-rate-limits","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/messages"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"meta","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":196000,"max_output_tokens":8000,"max_tokens":8000,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"minimax/MiniMax-M3":{"mode":"chat","base_model":"minimax-m3","supports_function_calling":true,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_system_messages":true,"supports_vision":true,"max_input_tokens":1000000,"max_output_tokens":128000,"provider":"minimax","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"mistral.devstral-2-123b":{"mode":"chat","base_model":"devstral-2-123b","max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"mistral/devstral-small-latest":{"mode":"chat","base_model":"devstral-small","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.mistral.ai/models/devstral-small-2-25-12","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"mistral/devstral-latest":{"mode":"chat","base_model":"devstral","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://mistral.ai/news/devstral-2-vibe-cli","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"mistral/devstral-medium-latest":{"mode":"chat","base_model":"devstral-medium","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://mistral.ai/news/devstral-2-vibe-cli","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"mistral/ministral-14b-2512":{"mode":"chat","base_model":"ministral-14b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://docs.mistral.ai/models/ministral-3-14b-25-12","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/ministral-14b-latest":{"mode":"chat","base_model":"ministral-14b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://docs.mistral.ai/models/ministral-3-14b-25-12","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/ministral-3b-2512":{"mode":"chat","base_model":"ministral-3b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.mistral.ai/models/ministral-3-3b-25-12","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"mistral/mistral-medium-3":{"mode":"chat","base_model":"mistral-medium-3","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-medium-3-5-26-04","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/voxtral-small-2507":{"mode":"chat","base_model":"voxtral-small","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.mistral.ai/models/voxtral-small-25-07","supports_audio_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"mistral/voxtral-small-latest":{"mode":"chat","base_model":"voxtral-small","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.mistral.ai/models/voxtral-small-25-07","supports_audio_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"mistral/zai-glm-5-2":{"mode":"chat","base_model":"glm-5-2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["none","minimal","low","medium","high","xhigh","max"],"source":"https://docs.mistral.ai/models/zai-glm-5-2","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"mistral/zai-glm-5-3":{"mode":"chat","base_model":"glm-5-3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://docs.mistral.ai/models/zai-glm-5-3","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"mistral/zai-glm-5":{"mode":"chat","base_model":"glm-5","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://docs.mistral.ai/models/zai-glm-5-3","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"mistral/zai-glm-latest":{"mode":"chat","base_model":"glm","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://docs.mistral.ai/models/zai-glm-5-3","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"mistral/glm-5-2":{"mode":"chat","base_model":"glm-5-2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["none","minimal","low","medium","high","xhigh","max"],"source":"https://docs.mistral.ai/models/zai-glm-5-2","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"mistral/mistral-large-2512":{"mode":"chat","base_model":"mistral-large","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://docs.mistral.ai/models/mistral-large-3-25-12","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/mistral-medium-2604":{"mode":"chat","base_model":"mistral-medium","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-medium-3-5-26-04","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/mistral-medium-3-5":{"mode":"chat","base_model":"mistral-medium-3-5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-medium-3-5-26-04","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/ministral-3-3b-2512":{"mode":"chat","base_model":"ministral-3-3b","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://mistral.ai/pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"mistral/ministral-3-8b-2512":{"mode":"chat","base_model":"ministral-3-8b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://mistral.ai/pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/ministral-3-14b-2512":{"mode":"chat","base_model":"ministral-3-14b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://mistral.ai/pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/ministral-8b-2512":{"mode":"chat","base_model":"ministral-8b","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://mistral.ai/pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"moonshot/kimi-k2.7-code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.kimi.ai/docs/pricing/chat-k27-code","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"moonshot","supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"model":"kimi-k2.7-code","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"moonshot/kimi-k2.6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.kimi.ai/docs/pricing/chat-k26","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"moonshot","supported_endpoints":["/v1/chat/completions","/v1/batch"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_prompt_caching":true,"supports_web_search":true,"model":"kimi-k2.6","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"moonshot/kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"reasoning_effort_levels":["low","high","max"],"source":"https://platform.kimi.ai/docs/pricing/chat-k3","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"moonshot","supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_prompt_caching":true,"supports_web_search":true,"model":"kimi-k3","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"nebius/deepseek-ai/DeepSeek-R1":{"mode":"chat","base_model":"deepseek-r1","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_reasoning":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/deepseek-ai/DeepSeek-R1-0528":{"mode":"chat","base_model":"deepseek-r1","max_tokens":164000,"max_input_tokens":164000,"max_output_tokens":164000,"supports_function_calling":true,"supports_reasoning":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":164000}}]},"nebius/deepseek-ai/DeepSeek-R1-Distill-Llama-70B":{"mode":"chat","base_model":"deepseek-r1-distill-llama-70b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/deepseek-ai/DeepSeek-V3":{"mode":"chat","base_model":"deepseek-v3","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/deepseek-ai/DeepSeek-V3-0324":{"mode":"chat","base_model":"deepseek-v3","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/google/gemma-3-27b-it":{"mode":"chat","base_model":"gemma-3-27b-it","max_tokens":110000,"max_input_tokens":110000,"max_output_tokens":110000,"supports_function_calling":true,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/google%2Fgemma-3-27b-it","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":110000}}]},"nebius/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"nebius/meta-llama/Llama-Guard-3-8B":{"mode":"chat","base_model":"llama-guard-3-8b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/meta-llama/Meta-Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/meta-llama/Meta-Llama-3.1-70B-Instruct":{"mode":"chat","base_model":"llama-3.1-70b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/meta-llama/Meta-Llama-3.1-405B-Instruct":{"mode":"chat","base_model":"llama-3.1-405b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/mistralai/Mistral-Nemo-Instruct-2407":{"mode":"chat","base_model":"mistral-nemo-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/NousResearch/Hermes-3-Llama-3.1-405B":{"mode":"chat","base_model":"hermes-3-llama-3.1-405b","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/nvidia/Llama-3.1-Nemotron-Ultra-253B-v1":{"mode":"chat","base_model":"llama-3.1-nemotron-ultra-253b-v1","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/nvidia/Llama-3.3-Nemotron-Super-49B-v1":{"mode":"chat","base_model":"llama-3.3-nemotron-super-49b-v1","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"nebius/Qwen/Qwen3-235B-A22B":{"mode":"chat","base_model":"qwen3-235b-a22b","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/Qwen/Qwen3-32B":{"mode":"chat","base_model":"qwen3-32b","max_tokens":40960,"max_input_tokens":40960,"max_output_tokens":40960,"supports_function_calling":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/Qwen%2FQwen3-32B","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"nebius/Qwen/Qwen3-30B-A3B":{"mode":"chat","base_model":"qwen3-30b-a3b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"nebius/Qwen/Qwen3-14B":{"mode":"chat","base_model":"qwen3-14b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"nebius/Qwen/Qwen3-4B":{"mode":"chat","base_model":"qwen3-4b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"nebius/Qwen/QwQ-32B":{"mode":"chat","base_model":"qwq-32b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"supports_reasoning":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"nebius/Qwen/Qwen2.5-72B-Instruct":{"mode":"chat","base_model":"qwen2.5-72b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/Qwen/Qwen2.5-32B-Instruct":{"mode":"chat","base_model":"qwen2.5-32b-instruct","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/Qwen/Qwen2.5-Coder-7B":{"mode":"chat","base_model":"qwen2.5-coder-7b","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"nebius/Qwen/Qwen2.5-VL-72B-Instruct":{"mode":"chat","base_model":"qwen2.5-vl-72b-instruct","max_tokens":32000,"max_input_tokens":32000,"max_output_tokens":32000,"supports_function_calling":true,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/image2text/Qwen%2FQwen2.5-VL-72B-Instruct","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"nebius/Qwen/Qwen2-VL-72B-Instruct":{"mode":"chat","base_model":"qwen2-vl-72b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_vision":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"nebius/Qwen/Qwen2-VL-7B-Instruct":{"mode":"chat","base_model":"qwen2-vl-7b-instruct","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_vision":true,"source":"https://nebius.com/prices","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"nebius/deepseek-ai/DeepSeek-V4-Flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_tokens":1048576,"max_input_tokens":1048576,"max_output_tokens":1048576,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/deepseek-ai%2FDeepSeek-V4-Flash","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"nebius/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_tokens":1024000,"max_input_tokens":1024000,"max_output_tokens":1024000,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/deepseek-ai%2FDeepSeek-V4-Flash-0731","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1024000}}]},"nebius/deepseek-ai/DeepSeek-V4-Pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_tokens":1048576,"max_input_tokens":1048576,"max_output_tokens":1048576,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/deepseek-ai%2FDeepSeek-V4-Pro","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"nebius/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","source":"https://tokenfactory.nebius.com/models/catalog/text2text/deepseek-ai%2FDeepSeek-V4-Pro-0813","supports_function_calling":true,"supports_reasoning":true,"provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"nebius/MiniMaxAI/MiniMax-M2.5":{"mode":"chat","base_model":"minimax-m2.5","max_tokens":196608,"max_input_tokens":196608,"max_output_tokens":196608,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/MiniMaxAI%2FMiniMax-M2.5","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":196608}}]},"nebius/MiniMaxAI/MiniMax-M3":{"mode":"chat","base_model":"minimax-m3","max_tokens":1048576,"max_input_tokens":1048576,"max_output_tokens":1048576,"supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/MiniMaxAI%2FMiniMax-M3","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"nebius/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"kimi-k2.6","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/image2text/moonshotai%2FKimi-K2.6","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/moonshotai/Kimi-K2.7-Code":{"mode":"chat","base_model":"kimi-k2.7-code","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/moonshotai%2FKimi-K2.7-Code","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/moonshotai/Kimi-K3":{"mode":"chat","base_model":"kimi-k3","max_tokens":1024000,"max_input_tokens":1024000,"max_output_tokens":1024000,"supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/image2text/moonshotai%2FKimi-K3","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1024000}}]},"nebius/NousResearch/Hermes-4-405B":{"mode":"chat","base_model":"hermes-4-405b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/NousResearch%2FHermes-4-405B","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"nebius/NousResearch/Hermes-4-70B":{"mode":"chat","base_model":"hermes-4-70b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/NousResearch%2FHermes-4-70B","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"nebius/nvidia/Cosmos3-Super-Reasoner":{"mode":"chat","base_model":"cosmos3-super-reasoner","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/image2text/nvidia%2FCosmos3-Super-Reasoner","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/nvidia/Llama-3_1-Nemotron-Ultra-253B-v1":{"mode":"chat","base_model":"llama-3-1-nemotron-ultra-253b-v1","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/nvidia%2FLlama-3_1-Nemotron-Ultra-253B-v1","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"nebius/nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B":{"mode":"chat","base_model":"nvidia-nemotron-3-nano-30b-a3b","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/nvidia%2FNVIDIA-Nemotron-3-Nano-30B-A3B","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/nvidia/Nemotron-3-Nano-Omni":{"mode":"chat","base_model":"nemotron-3-nano-omni","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/nvidia%2FNemotron-3-Nano-Omni","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/nvidia/nemotron-3-super-120b-a12b":{"mode":"chat","base_model":"nemotron-3-super-120b-a12b","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/nvidia%2Fnemotron-3-super-120b-a12b","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/nvidia/Nemotron-3-Ultra-550b-a55b":{"mode":"chat","base_model":"nemotron-3-ultra-550b-a55b","max_tokens":1048576,"max_input_tokens":1048576,"max_output_tokens":1048576,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/nvidia%2FNemotron-3-Ultra-550b-a55b","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"nebius/nvidia/Nemotron-3_5-Lightning":{"mode":"chat","base_model":"nemotron-3-5-lightning","max_tokens":1048576,"max_input_tokens":1048576,"max_output_tokens":1048576,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/nvidia%2FNemotron-3_5-Lightning","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"nebius/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/openai%2Fgpt-oss-120b","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"nebius/openbmb/MiniCPM-V-4_5":{"mode":"chat","base_model":"openbmb/minicpm-v-4-5","max_tokens":32000,"max_input_tokens":32000,"max_output_tokens":32000,"supports_vision":true,"source":"https://tokenfactory.nebius.com/models/catalog/image2text/openbmb%2FMiniCPM-V-4_5","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"nebius/Qwen/Qwen3-235B-A22B-Instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/Qwen%2FQwen3-235B-A22B-Instruct-2507","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/Qwen/Qwen3-30B-A3B-Instruct-2507":{"mode":"chat","base_model":"qwen3-30b-a3b-instruct","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/Qwen%2FQwen3-30B-A3B-Instruct-2507","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/Qwen/Qwen3-Next-80B-A3B-Thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/Qwen%2FQwen3-Next-80B-A3B-Thinking","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"nebius/Qwen/Qwen3.5-397B-A17B":{"mode":"chat","base_model":"qwen3.5-397b-a17b","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/Qwen%2FQwen3.5-397B-A17B","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"nebius/zai-org/GLM-5.1":{"mode":"chat","base_model":"glm-5.1","max_tokens":202752,"max_input_tokens":202752,"max_output_tokens":202752,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/zai-org%2FGLM-5.1","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"nebius/zai-org/GLM-5.2":{"mode":"chat","base_model":"glm-5.2","max_tokens":1048576,"max_input_tokens":1048576,"max_output_tokens":1048576,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/zai-org%2FGLM-5.2","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"nebius/zai-org/GLM-5.3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/zai-org%2FGLM-5.3","supports_function_calling":true,"supports_reasoning":true,"provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"nebius/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"glm-5.3-flash","max_tokens":1024000,"max_input_tokens":1024000,"max_output_tokens":1024000,"supports_function_calling":true,"supports_reasoning":true,"source":"https://tokenfactory.nebius.com/models/catalog/text2text/zai-org%2FGLM-5.3-Flash","provider":"nebius","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1024000}}]},"nvidia.nemotron-super-3-120b":{"mode":"chat","base_model":"nvidia-nemotron-super-3-120b","max_input_tokens":256000,"max_output_tokens":32768,"max_tokens":32768,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"bedrock","supported_endpoints":["/v1/chat/completions","/v1/batch"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"oci/meta.llama-3.1-8b-instruct":{"mode":"chat","base_model":"llama-3.1-8b-instruct","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/meta.llama-3.1-70b-instruct":{"mode":"chat","base_model":"llama-3.1-70b-instruct","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/google.gemini-2.5-flash":{"mode":"chat","base_model":"gemini-2.5-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_vision":true,"supports_native_streaming":true,"supports_image_size":false,"supports_reasoning":true,"supports_system_messages":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_video_input":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"oci/google.gemini-2.5-pro":{"mode":"chat","base_model":"gemini-2.5-pro","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_vision":true,"supports_native_streaming":true,"supports_reasoning":true,"supports_system_messages":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_video_input":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"oci/google.gemini-2.5-flash-lite":{"mode":"chat","base_model":"gemini-2.5-flash-lite","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_vision":true,"supports_native_streaming":true,"supports_image_size":false,"supports_reasoning":false,"supports_system_messages":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_video_input":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"oci/cohere.command-a-vision":{"mode":"chat","base_model":"command-a-vision","max_input_tokens":256000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.oracle.com/artificial-intelligence/enterprise-ai/cost-estimator/","supports_function_calling":true,"supports_response_schema":false,"supports_native_streaming":true,"supports_vision":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"oci/cohere.command-a-reasoning":{"mode":"chat","base_model":"command-a","max_input_tokens":256000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://www.oracle.com/artificial-intelligence/enterprise-ai/cost-estimator/","supports_function_calling":false,"supports_response_schema":false,"supports_native_streaming":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"oci/cohere.command-a-reasoning-08-2025":{"mode":"chat","base_model":"command-a","max_input_tokens":256000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/cohere.command-a-vision-07-2025":{"mode":"chat","base_model":"command-a-vision","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_vision":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/cohere.command-a-translate-08-2025":{"mode":"chat","base_model":"command-a-translate","max_input_tokens":256000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":false,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/cohere.command-r-08-2024":{"mode":"chat","base_model":"command-r","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/cohere.command-r-plus-08-2024":{"mode":"chat","base_model":"command-r-plus","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/meta.llama-3.2-11b-vision-instruct":{"mode":"chat","base_model":"llama-3.2-11b-vision-instruct","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"supports_vision":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/meta.llama-3.3-70b-instruct-fp8-dynamic":{"mode":"chat","base_model":"llama-3.3-70b-instruct-fp8-dynamic","max_input_tokens":128000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"oci/xai.grok-4-fast":{"mode":"chat","base_model":"grok-4-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"oci/xai.grok-4.1-fast":{"mode":"chat","base_model":"grok-4.1-fast","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"oci/xai.grok-4.20":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"oci/xai.grok-4.20-multi-agent":{"mode":"chat","base_model":"grok-4.20-multi-agent","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"oci/xai.grok-code-fast-1":{"mode":"chat","base_model":"grok-code-fast-1","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_response_schema":false,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"oci/openai.gpt-5":{"mode":"chat","base_model":"gpt-5","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_native_streaming":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"oci/openai.gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_native_streaming":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"oci/openai.gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/pricing","supports_function_calling":true,"supports_native_streaming":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"oci","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-sonnet-4.6":{"mode":"chat","base_model":"claude-sonnet-4.6","supports_adaptive_thinking":true,"supports_legacy_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_max_reasoning_effort":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"supports_audio_input":false,"supports_pdf_input":true,"supports_response_schema":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-4.6":{"mode":"chat","base_model":"claude-opus-4.6","supports_adaptive_thinking":true,"supports_legacy_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_max_reasoning_effort":true,"supports_tool_choice":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"supports_response_schema":true,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_pdf_input":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-4.7":{"mode":"chat","base_model":"claude-opus-4.7","supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_max_reasoning_effort":true,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"prompt_cache_min_tokens":2048,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","prompt_cache_min_tokens":512,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_assistant_prefill":false,"supports_audio_input":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_max_reasoning_effort":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_xhigh_reasoning_effort":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-5.5":{"mode":"chat","base_model":"claude-opus-5.5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/deepseek/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1024000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_pdf_input":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"openrouter/deepseek/deepseek-v4.1-flash":{"mode":"chat","base_model":"deepseek-v4.1-flash","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"off_peak_pricing":{"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"cache_read_input_token_cost":3e-9},"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/deepseek/deepseek-v4-pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro-0813","max_input_tokens":1048576,"max_output_tokens":384000,"max_tokens":384000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"off_peak_pricing":{"windows":[{"weekdays":["saturday","sunday"],"hours_utc":"00:00-00:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"00:00-01:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"04:00-06:00"},{"weekdays":["monday","tuesday","wednesday","thursday","friday"],"hours_utc":"10:00-00:00"}],"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"cache_read_input_token_cost":2.2e-8},"supports_audio_input":false,"supports_pdf_input":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"openrouter/google/gemini-3.1-flash-lite-preview":{"mode":"chat","base_model":"gemini-3.1-flash-lite-preview","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":2000,"source":"https://openrouter.ai/api/v1/models","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"tpm":800000,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.1-flash-lite":{"mode":"chat","base_model":"gemini-3.1-flash-lite","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":2000,"source":"https://openrouter.ai/api/v1/models","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"tpm":800000,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.1-pro-preview":{"mode":"chat","base_model":"gemini-3.1-pro-preview","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/nvidia/nemotron-3.5-lightning":{"mode":"chat","base_model":"nemotron-3.5-lightning","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/openai/gpt-5.1-codex-max":{"mode":"chat","base_model":"gpt-5.1-codex-max","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","default_reasoning_effort":"medium","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"source":"https://openrouter.ai/api/v1/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-sol-pro":{"mode":"chat","base_model":"gpt-5.6-sol-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/qwen/qwen3-coder-plus":{"mode":"chat","base_model":"qwen3-coder-plus","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.6-plus":{"mode":"chat","base_model":"qwen3.6-plus","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.5-35b-a3b":{"mode":"chat","base_model":"qwen3.5-35b-a3b","max_input_tokens":256000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_audio_input":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/qwen/qwen3.5-27b":{"mode":"chat","base_model":"qwen3.5-27b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.5-122b-a10b":{"mode":"chat","base_model":"qwen3.5-122b-a10b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.5-flash-02-23":{"mode":"chat","base_model":"qwen3.5-flash-02-23","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.5-plus-02-15":{"mode":"chat","base_model":"qwen3.5-plus-02-15","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.5-397b-a17b":{"mode":"chat","base_model":"qwen3.5-397b-a17b","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/xiaomi/mimo-v2.5-pro":{"mode":"chat","base_model":"mimo-v2.5-pro","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"supports_response_schema":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/xiaomi/mimo-v2.5":{"mode":"chat","base_model":"mimo-v2.5","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":true,"supports_audio_input":true,"supports_pdf_input":false,"supports_video_input":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/z-ai/glm-5":{"mode":"chat","base_model":"glm-5","max_input_tokens":198000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/z-ai/glm-5.1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/minimax/minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"supports_prompt_caching":true,"supports_computer_use":false,"supports_pdf_input":false,"supports_response_schema":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openrouter/free":{"mode":"chat","base_model":"free","max_input_tokens":200000,"max_tokens":200000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":200000}}]},"openrouter/openrouter/bodybuilder":{"mode":"chat","base_model":"bodybuilder","max_input_tokens":128000,"max_tokens":128000,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"perplexity/preset/fast-search":{"mode":"responses","base_model":"preset/fast-search","supports_web_search":true,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/preset/deep-research":{"mode":"responses","base_model":"preset/deep-research","supports_web_search":true,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/preset/advanced-deep-research":{"mode":"responses","base_model":"preset/advanced","supports_web_search":true,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5.1":{"mode":"responses","base_model":"gpt-5.1","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5-mini":{"mode":"responses","base_model":"gpt-5-mini","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-opus-4-6":{"mode":"responses","base_model":"claude-opus-4-6","supports_adaptive_thinking":true,"supports_legacy_thinking":true,"supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"supports_output_config":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-opus-4-7":{"mode":"responses","base_model":"claude-opus-4-7","supports_adaptive_thinking":true,"supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"supports_output_config":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-opus-4-5":{"mode":"responses","base_model":"claude-opus-4-5","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"supports_output_config":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-sonnet-4-5":{"mode":"responses","base_model":"claude-sonnet-4-5","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-haiku-4-5":{"mode":"responses","base_model":"claude-haiku-4-5","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-3-pro-preview":{"mode":"responses","base_model":"gemini-3-pro","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-3-flash-preview":{"mode":"responses","base_model":"gemini-3-flash","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-2.5-pro":{"mode":"responses","base_model":"gemini-2.5-pro","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-2.5-flash":{"mode":"responses","base_model":"gemini-2.5-flash","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"supports_image_size":false,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/xai/grok-4-1-fast-non-reasoning":{"mode":"responses","base_model":"grok-4-1-fast-non","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/perplexity/sonar":{"mode":"chat","base_model":"sonar","max_input_tokens":128000,"max_tokens":128000,"supports_web_search":true,"provider":"perplexity","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"perplexity/perplexity/deepseek-v4-flash-0731":{"mode":"responses","base_model":"deepseek-v4-flash","source":"https://docs.perplexity.ai/docs/agent-api/models","supports_web_search":true,"supports_reasoning":true,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/perplexity/glm-5.2":{"mode":"responses","base_model":"glm-5.2","source":"https://docs.perplexity.ai/docs/agent-api/models","supports_web_search":true,"supports_reasoning":true,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/perplexity/kimi-k3":{"mode":"responses","base_model":"kimi-k3","reasoning_effort_levels":["minimal","low","medium","high","xhigh","max"],"source":"https://docs.perplexity.ai/docs/agent-api/models","supports_web_search":true,"supports_reasoning":true,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/perplexity/kimi-k2.7-code":{"mode":"responses","base_model":"kimi-k2.7-code","source":"https://docs.perplexity.ai/docs/agent-api/models","supports_web_search":true,"supports_reasoning":false,"supports_function_calling":true,"provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"bedrock/ap-northeast-1/qwen.qwen3-next-80b-a3b":{"mode":"chat","base_model":"qwen3-next-80b-a3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/ap-south-1/qwen.qwen3-next-80b-a3b":{"mode":"chat","base_model":"qwen3-next-80b-a3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/ap-southeast-2/qwen.qwen3-next-80b-a3b":{"mode":"chat","base_model":"qwen3-next-80b-a3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/eu-west-1/qwen.qwen3-next-80b-a3b":{"mode":"chat","base_model":"qwen3-next-80b-a3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/eu-west-2/qwen.qwen3-next-80b-a3b":{"mode":"chat","base_model":"qwen3-next-80b-a3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/sa-east-1/qwen.qwen3-next-80b-a3b":{"mode":"chat","base_model":"qwen3-next-80b-a3b","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"replicate/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","supports_function_calling":true,"supports_system_messages":true,"provider":"replicate","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"sambanova/MiniMax-M2.7":{"mode":"chat","base_model":"minimax-m2.7","max_input_tokens":196608,"max_output_tokens":131072,"max_tokens":131072,"source":"https://cloud.sambanova.ai/plans/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"sambanova","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"sambanova/DeepSeek-V3.2":{"mode":"chat","base_model":"deepseek-v3.2","max_tokens":32768,"max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"sambanova/gemma-4-31B-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_vision":true,"source":"https://cloud.sambanova.ai/plans/pricing","provider":"sambanova","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"scx-ai/GLM-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://scx.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"scx-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"scx-ai/Qwen3.8-Max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://scx.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"scx-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"together_ai/MiniMaxAI/MiniMax-M3":{"mode":"chat","base_model":"minimax-m3","max_input_tokens":524288,"max_tokens":524288,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"together_ai/Prism-ML/Ternary-Bonsai-27B":{"mode":"chat","base_model":"prism-ml/ternary-bonsai-27b","max_input_tokens":262144,"max_tokens":262144,"source":"https://docs.together.ai/docs/serverless-models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"together_ai/Qwen/Qwen3.5-9B":{"mode":"chat","base_model":"qwen3.5-9b","max_input_tokens":262144,"max_tokens":262144,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"together_ai/Qwen/Qwen3.6-Plus":{"mode":"chat","base_model":"qwen3.6-plus","max_input_tokens":1000000,"max_tokens":1000000,"source":"https://api.together.ai/v1/models","supports_reasoning":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"together_ai/Qwen/Qwen3.7-Max":{"mode":"chat","base_model":"qwen3.7-max","max_input_tokens":1000000,"max_tokens":1000000,"source":"https://api.together.ai/v1/models","supports_prompt_caching":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"together_ai/Qwen/Qwen3.7-Plus":{"mode":"chat","base_model":"qwen3.7-plus","max_input_tokens":1000000,"max_tokens":1000000,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"together_ai/Qwen/Qwen3.8-2.4T-A95B":{"mode":"chat","base_model":"qwen3.8-2.4t-a95b","max_input_tokens":1010000,"max_tokens":1010000,"source":"https://api.together.ai/v1/models","supports_prompt_caching":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1010000}}]},"together_ai/arize-ai/qwen-2-1.5b-instruct":{"mode":"chat","base_model":"arize-ai/qwen-2-1.5b-instruct","max_input_tokens":32768,"max_tokens":32768,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"together_ai/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","deprecation_date":"2026-09-29","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"together_ai/deepseek-ai/DeepSeek-V4.1-Flash":{"mode":"chat","base_model":"deepseek-v4.1-flash","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","deprecation_date":"2026-09-29","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"together_ai/meta-models/Muse-Glimmer-30B":{"mode":"chat","base_model":"models/muse-glimmer-30b","max_input_tokens":131072,"max_tokens":131072,"source":"https://api.together.ai/v1/models","supports_prompt_caching":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"together_ai/moonshotai/Kimi-K3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_tokens":1048576,"reasoning_effort_levels":["low","high","max"],"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"together_ai/thinkingmachines/Inkling":{"mode":"chat","base_model":"thinkingmachines/inkling","max_input_tokens":524288,"max_tokens":524288,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"together_ai/zai-org/GLM-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048575,"max_output_tokens":128000,"max_tokens":128000,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"together_ai/zai-org/GLM-5.3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048575,"max_output_tokens":128000,"max_tokens":128000,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"together_ai/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1048575,"max_output_tokens":128000,"max_tokens":128000,"source":"https://api.together.ai/v1/models","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us-gov.anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us-gov.anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"thinking_always_on":true,"supports_forced_tool_use":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us-gov.anthropic.claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us-gov.nvidia.nemotron-nano-3-30b":{"mode":"chat","base_model":"nvidia-nemotron-nano-3-30b","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"us-gov.nvidia.nemotron-nano-12b-v2":{"mode":"chat","base_model":"nvidia-nemotron-nano-12b-v2","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_system_messages":true,"supports_vision":true,"supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"us-gov.nvidia.nemotron-nano-9b-v2":{"mode":"chat","base_model":"nvidia-nemotron-nano-9b-v2","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supports_system_messages":true,"supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"us-gov.nvidia.nemotron-super-3-120b":{"mode":"chat","base_model":"nvidia-nemotron-super-3-120b","max_input_tokens":256000,"max_output_tokens":32768,"max_tokens":32768,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"us-gov.openai.gpt-oss-20b-1:0":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us-gov.openai.gpt-oss-120b-1:0":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us-gov.xai.grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"vertex_ai/claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","deprecation_date":"2026-10-15","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"regional_endpoint_uplift_multiplier":1.1,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supports_assistant_prefill":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_native_streaming":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"vertex_ai/claude-opus-4-6@default":{"mode":"chat","base_model":"claude-opus-4-6","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"vertex_ai","supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","deprecation_date":"2027-06-08","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-fable-5@default":{"mode":"chat","base_model":"claude-fable-5","deprecation_date":"2027-06-08","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","deprecation_date":"2027-01-24","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":1024,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","tool_use_system_prompt_tokens":346,"supports_minimal_reasoning_effort":true,"vertex_multi_region_only":true,"supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-opus-5@default":{"mode":"chat","base_model":"claude-opus-5","deprecation_date":"2027-01-24","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","provider_specific_entry":{"us":1.1},"supports_output_config":true,"tool_use_system_prompt_tokens":286,"supports_web_search":true,"supports_forced_tool_choice":false,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-opus-5-5@default":{"mode":"chat","base_model":"claude-opus-5-5","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_native_structured_output":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":512,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","deprecation_date":"2027-05-28","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":1000000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":1024,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","tool_use_system_prompt_tokens":346,"supports_minimal_reasoning_effort":true,"vertex_multi_region_only":true,"supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-opus-4-8@default":{"mode":"chat","base_model":"claude-opus-4-8","deprecation_date":"2027-05-28","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"supports_adaptive_thinking":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":1024,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","deprecation_date":"2026-12-24","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":1024,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"vertex_ai","supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"vertex_ai/gemini-3-pro-image":{"mode":"image_generation","base_model":"gemini-3-pro-image","deprecation_date":"2027-05-28","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"vertex_ai/gemini-3.1-flash-image":{"mode":"image_generation","base_model":"gemini-3.1-flash-image","deprecation_date":"2027-05-28","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"vertex_ai/gemini-3.1-flash-image-preview":{"mode":"image_generation","base_model":"gemini-3.1-flash-image","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"vertex_ai/gemini-3.1-flash-lite-image":{"mode":"image_generation","base_model":"gemini-3.1-flash-lite-image","max_input_tokens":65536,"max_output_tokens":4096,"max_tokens":4096,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text","image"],"supports_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":false,"supports_system_messages":true,"supports_video_input":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"vertex_ai/gemini-3.1-flash-lite-preview":{"mode":"chat","base_model":"gemini-3.1-flash-lite","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://cloud.google.com/vertex-ai/generative-ai/pricing#gemini-models","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"supports_native_streaming":true,"web_search_billing_unit":"per_query","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"vertex_ai/google/gemma-4-26b-a4b-it-maas":{"mode":"chat","base_model":"gemma-4","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supported_regions":["global"],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","tpm":250000,"rpm":10,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"wandb/moonshotai/Kimi-K2.5":{"mode":"chat","base_model":"kimi-k2.5","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"source":"https://wandb.ai/inference/coreweave/cw_moonshotai_Kimi-K2.5","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"watsonx/bigscience/mt0-xxl":{"mode":"chat","base_model":"mt0-xxl","max_tokens":4096,"max_input_tokens":4096,"max_output_tokens":4096,"supports_function_calling":false,"supports_parallel_function_calling":false,"supports_vision":false,"source":"https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models.html?context=wx","provider":"watsonx","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"watsonx/meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"mode":"chat","base_model":"llama-4-maverick-17b-128e-instruct-fp8","max_tokens":8192,"max_input_tokens":131072,"max_output_tokens":8192,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":false,"source":"https://dataplatform.cloud.ibm.com/docs/content/wsj/analyze-data/fm-models.html?context=wx","provider":"watsonx","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"xai/grok-4.20-multi-agent-beta-0309":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"supported_endpoints":["/v1/responses"],"provider":"xai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta-0309-reasoning":{"mode":"chat","base_model":"grok-4.20-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-0309-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_reasoning":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta-0309-non-reasoning":{"mode":"chat","base_model":"grok-4.20-beta-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.3":{"mode":"chat","base_model":"grok-4.3","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.3-latest":{"mode":"chat","base_model":"grok-4.3","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.5":{"mode":"chat","base_model":"grok-4.5","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"source":"https://docs.x.ai/developers/models/grok-4.5","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"xai/grok-4.5-latest":{"mode":"chat","base_model":"grok-4.5","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"source":"https://docs.x.ai/developers/models/grok-4.5","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"xai/grok-build-latest":{"mode":"chat","base_model":"grok-build","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"xai/grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"source":"https://docs.x.ai/developers/models/grok-4.6","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"xai/grok-4.7":{"mode":"chat","base_model":"grok-4.7","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"source":"https://docs.x.ai/developers/models/grok-4.7","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"xai","regional_endpoint_uplift_multiplier":1.1,"web_search_billing_unit":"per_query","default_reasoning_effort":"high","reasoning_effort_levels":["low","medium","high","xhigh"],"supports_low_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_none_reasoning_effort":false,"thinking_always_on":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_code_execution":true,"supports_service_tier":true,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_regions":["us-east-1","us-west-2","us-central-1"],"rpm":9000,"tpm":50000000,"metadata":{"notes":"Released 2026-09-21. Knowledge cutoff May 2026. Long-context rates apply to all tokens in a request once the prompt reaches 200k tokens (>= 200k, not strictly above). No documented text output limit; max_output_tokens mirrors the 500k context window. Batch API not supported. rpm derived from 150 requests/second. US regional endpoint (us.api.x.ai) billed at 1.1x. Grok 4.7 Fast (2x rates) is Cursor/Grok Build only and not on the public API, so it has no entry.","calculation":"Priority tier = 2x standard rates on all token types, applied after caching discount. Web search = $5 per 1k calls = $0.005 per call."},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"zai.glm-5":{"mode":"chat","base_model":"zai.glm-5","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_reasoning":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"zai.glm-4.7-flash":{"mode":"chat","base_model":"zai.glm-4.7-flash","max_input_tokens":203000,"max_output_tokens":4000,"max_tokens":4000,"supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"source":"https://aws.amazon.com/bedrock/pricing/","supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"zai/glm-5":{"mode":"chat","base_model":"glm-5","max_input_tokens":200000,"max_output_tokens":128000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"zai/glm-5.3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1000000,"max_output_tokens":128000,"source":"https://docs.z.ai/guides/overview/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"zai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"zai/glm-5.3-flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1048576,"max_output_tokens":128000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","supports_vision":true,"provider":"zai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"zai/glm-5.1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":200000,"max_output_tokens":128000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"zai/glm-5-code":{"mode":"chat","base_model":"glm-5-code","max_input_tokens":200000,"max_output_tokens":128000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"zai/glm-4.7-flash":{"mode":"chat","base_model":"glm-4.7-flash","max_input_tokens":200000,"max_output_tokens":128000,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"source":"https://docs.z.ai/guides/overview/pricing","provider":"zai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"fireworks_ai/accounts/fireworks/models/qwen3p7-plus":{"mode":"chat","base_model":"qwen3.7-plus","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"fireworks_ai/accounts/fireworks/routers/glm-5p1-fast":{"mode":"chat","base_model":"accounts/fireworks/routers/glm-5.1-fast","max_input_tokens":202800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/routers/kimi-k2p6-fast":{"mode":"chat","base_model":"accounts/fireworks/routers/kimi-k2.6-fast","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast":{"mode":"chat","base_model":"accounts/fireworks/routers/kimi-k2.7-code-fast","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"scaleway/qwen/qwen3.5-397b-a17b":{"mode":"chat","base_model":"qwen3.5-397b-a17b","max_input_tokens":256000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"scaleway/qwen/qwen3.6-35b-a3b":{"mode":"chat","base_model":"qwen3.6-35b-a3b","max_input_tokens":256000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_reasoning":true,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"scaleway/qwen/qwen3-235b-a22b-instruct-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-instruct","max_input_tokens":256000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"scaleway/qwen/qwen3-coder-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-coder-30b-a3b-instruct","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"deprecation_date":"2026-10-01","provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"scaleway/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"scaleway/google/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"gemma-4-26b-a4b-it","max_input_tokens":256000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"scaleway/mistralai/mistral-medium-3.5-128b":{"mode":"chat","base_model":"mistral-medium-3.5-128b","max_input_tokens":256000,"max_output_tokens":16384,"max_tokens":16384,"supports_reasoning":true,"supports_function_calling":true,"supports_vision":true,"supports_tool_choice":true,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"scaleway/mistralai/mistral-small-3.2-24b-instruct-2506":{"mode":"chat","base_model":"mistral-small-3.2-24b-instruct","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_vision":true,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"scaleway/mistralai/pixtral-12b-2409":{"mode":"chat","base_model":"pixtral-12b","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"supports_vision":true,"supports_function_calling":true,"deprecation_date":"2026-10-01","provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"scaleway/meta/llama-3.3-70b-instruct":{"mode":"chat","base_model":"llama-3.3-70b-instruct","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"libertai/hermes-3-8b-tee":{"mode":"chat","base_model":"hermes-3-8b-tee","max_tokens":16000,"max_input_tokens":16000,"max_output_tokens":16000,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":false,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16000}}]},"libertai/gemma-4-31b-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"libertai/gemma-4-31b-it-thinking":{"mode":"chat","base_model":"gemma-4-31b-it-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":true,"supports_reasoning":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"libertai/qwen3.6-27b":{"mode":"chat","base_model":"qwen3.6-27b","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"libertai/qwen3.6-27b-thinking":{"mode":"chat","base_model":"qwen3.6-27b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":true,"supports_reasoning":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"libertai/qwen3.6-35b-a3b":{"mode":"chat","base_model":"qwen3.6-35b-a3b","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"libertai/qwen3.6-35b-a3b-thinking":{"mode":"chat","base_model":"qwen3.6-35b-a3b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":true,"supports_reasoning":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"libertai/qwen3.5-122b-a10b":{"mode":"chat","base_model":"qwen3.5-122b-a10b","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"libertai/qwen3.5-122b-a10b-thinking":{"mode":"chat","base_model":"qwen3.5-122b-a10b-thinking","max_tokens":262144,"max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":true,"supports_reasoning":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"libertai/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_tokens":200000,"max_input_tokens":200000,"max_output_tokens":200000,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":false,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":200000}}]},"libertai/deepseek-v4-flash-thinking":{"mode":"chat","base_model":"deepseek-v4-flash-thinking","max_tokens":200000,"max_input_tokens":200000,"max_output_tokens":200000,"supports_function_calling":true,"supports_tool_choice":true,"supports_system_messages":true,"supports_vision":false,"supports_reasoning":true,"source":"https://docs.libertai.io/apis/text/","provider":"libertai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":200000}}]},"vertex_ai/claude-sonnet-5@default":{"mode":"chat","base_model":"claude-sonnet-5","deprecation_date":"2026-12-24","regional_endpoint_uplift_multiplier":1.1,"supports_mid_conversation_system":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"prompt_cache_min_tokens":1024,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/claude-sonnet-4-6@default":{"mode":"chat","base_model":"claude-sonnet-4-6","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"tool_use_system_prompt_tokens":346,"provider":"vertex_ai","supports_web_search":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"bedrock_mantle/openai.gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"bedrock_mantle/openai.gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"bedrock_mantle/openai.gpt-oss-safeguard-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"bedrock_mantle/openai.gpt-oss-safeguard-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"bedrock_mantle/openai.gpt-5.6-sol":{"mode":"responses","base_model":"gpt-5.6-sol","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_web_search":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/openai.gpt-5.6-terra":{"mode":"responses","base_model":"gpt-5.6-terra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_web_search":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/openai.gpt-5.6-cyber":{"mode":"responses","base_model":"gpt-5.6-cyber","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/openai.gpt-daybreak-blue-5.6-sol":{"mode":"responses","base_model":"gpt-daybreak-blue-5.6-sol","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"source":"https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-daybreak-blue-56-sol.html","provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/openai.gpt-5.6-luna":{"mode":"responses","base_model":"gpt-5.6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_web_search":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us.openai.gpt-5.6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"supports_sampling_params":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"global.openai.gpt-5.6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"supports_sampling_params":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us.openai.gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"supports_sampling_params":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"global.openai.gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"supports_sampling_params":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"us.openai.gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"supports_sampling_params":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"global.openai.gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"supports_sampling_params":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/openai.gpt-6-astra":{"mode":"responses","base_model":"gpt-6-astra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_none_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"source":"https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html","provider":"bedrock_mantle","supports_prompt_cache_breakpoints":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"global.openai.gpt-6-astra":{"mode":"responses","base_model":"gpt-6-astra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_minimal_reasoning_effort":false,"supports_none_reasoning_effort":false,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"source":"https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-astra.html","provider":"bedrock","supports_prompt_cache_breakpoints":true,"supported_endpoints":["/v1/responses"],"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/openai.gpt-5.5":{"mode":"responses","base_model":"gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_web_search":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/openai.gpt-5.4":{"mode":"responses","base_model":"gpt-5.4","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_web_search":true,"provider":"bedrock_mantle","supports_service_tier":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/google.gemma-4-31b":{"mode":"chat","base_model":"gemma-4-31b","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"bedrock_mantle/google.gemma-4-26b-a4b":{"mode":"chat","base_model":"gemma-4-26b-a4b","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"bedrock_mantle/google.gemma-4-e2b":{"mode":"chat","base_model":"gemma-4-e2b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":true,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/anthropic.claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","supports_tool_search":true,"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_assistant_prefill":true,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_native_structured_output":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":4096,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"volcengine/doubao-seed-2-0-pro-260215":{"mode":"chat","base_model":"doubao-seed-2-0-pro","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://docs.volcengine.com/docs/82379/1330310","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":4.6e-7,"output_cost_per_token":0.0000023,"range":[0,32000]},{"input_cost_per_token":7e-7,"output_cost_per_token":0.0000035,"range":[32000,128000]},{"input_cost_per_token":0.0000014,"output_cost_per_token":0.000007,"range":[128000,256000]}],"provider":"volcengine","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"volcengine/doubao-seed-2-1-pro-260628":{"mode":"chat","base_model":"doubao-seed-2-1-pro","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://www.volcengine.com/docs/82379/1544106","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"provider":"volcengine","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"volcengine/doubao-seed-2-1-turbo-260628":{"mode":"chat","base_model":"doubao-seed-2-1-turbo","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://www.volcengine.com/docs/82379/1544106","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"provider":"volcengine","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"volcengine/doubao-seed-2-0-lite-260215":{"mode":"chat","base_model":"doubao-seed-2-0-lite","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://docs.volcengine.com/docs/82379/1330310","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":8.7e-8,"output_cost_per_token":5.2e-7,"range":[0,32000]},{"input_cost_per_token":1.3e-7,"output_cost_per_token":7.8e-7,"range":[32000,128000]},{"input_cost_per_token":2.6e-7,"output_cost_per_token":0.0000016,"range":[128000,256000]}],"provider":"volcengine","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"volcengine/doubao-seed-2-0-mini-260215":{"mode":"chat","base_model":"doubao-seed-2-0-mini","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://docs.volcengine.com/docs/82379/1330310","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":2.9e-8,"output_cost_per_token":2.9e-7,"range":[0,32000]},{"input_cost_per_token":5.8e-8,"output_cost_per_token":5.8e-7,"range":[32000,128000]},{"input_cost_per_token":1.2e-7,"output_cost_per_token":0.0000012,"range":[128000,256000]}],"provider":"volcengine","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"volcengine/doubao-seed-2-0-code-preview-260215":{"mode":"chat","base_model":"doubao-seed-2-0-code","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://docs.volcengine.com/docs/82379/1330310","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"tiered_pricing":[{"input_cost_per_token":4.6e-7,"output_cost_per_token":0.0000023,"range":[0,32000]},{"input_cost_per_token":7e-7,"output_cost_per_token":0.0000035,"range":[32000,128000]},{"input_cost_per_token":0.0000014,"output_cost_per_token":0.000007,"range":[128000,256000]}],"provider":"volcengine","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/us-east-1/zai.glm-5":{"mode":"chat","base_model":"zai.glm-5","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/us-west-2/zai.glm-5":{"mode":"chat","base_model":"zai.glm-5","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_native_structured_output":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"snowflake/claude-sonnet-4-5":{"mode":"chat","base_model":"claude-sonnet-4-5","max_tokens":16384,"max_input_tokens":200000,"max_output_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_response_schema":true,"prompt_cache_min_tokens":1024,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","supports_adaptive_thinking":true,"supports_legacy_thinking":true,"max_tokens":16384,"max_input_tokens":200000,"max_output_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_response_schema":true,"prompt_cache_min_tokens":1024,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/claude-4-sonnet":{"mode":"chat","base_model":"claude-sonnet-4","max_tokens":16384,"max_input_tokens":200000,"max_output_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_response_schema":true,"prompt_cache_min_tokens":1024,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/claude-4-opus":{"mode":"chat","base_model":"claude-opus-4","max_tokens":16384,"max_input_tokens":200000,"max_output_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"supports_response_schema":true,"prompt_cache_min_tokens":1024,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","max_tokens":16384,"max_input_tokens":200000,"max_output_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_response_schema":true,"prompt_cache_min_tokens":4096,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/claude-3-7-sonnet":{"mode":"chat","base_model":"claude-3-7-sonnet","max_tokens":16384,"max_input_tokens":200000,"max_output_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/openai-gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","max_tokens":16384,"max_input_tokens":300000,"max_output_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/openai-gpt-5":{"mode":"chat","base_model":"gpt-5","max_tokens":16384,"max_input_tokens":300000,"max_output_tokens":16384,"supports_function_calling":true,"supports_vision":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/openai-gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","max_tokens":16384,"max_input_tokens":1000000,"max_output_tokens":16384,"supports_function_calling":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/openai-gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","max_tokens":16384,"max_input_tokens":5000000,"max_output_tokens":16384,"supports_function_calling":true,"supports_system_messages":true,"supports_response_schema":true,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"snowflake/llama4-maverick":{"mode":"chat","base_model":"llama-4-maverick","max_tokens":16384,"max_input_tokens":128000,"max_output_tokens":16384,"supports_function_calling":true,"supports_system_messages":true,"provider":"snowflake","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"tensormesh/Qwen/Qwen3.5-397B-A17B-FP8":{"mode":"chat","base_model":"qwen3.5-397b-a17b-fp8","max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"tensormesh/Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct-fp8","max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"tensormesh/Qwen/Qwen3.6-27B-FP8":{"mode":"chat","base_model":"qwen3.6-27b-fp8","max_input_tokens":262144,"max_output_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"tensormesh/lukealonso/GLM-5.1-NVFP4-MTP":{"mode":"chat","base_model":"lukealonso/glm-5.1-nvfp4-mtp","max_input_tokens":202752,"max_output_tokens":202752,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"tensormesh/deepseek-ai/DeepSeek-V4-Flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"tensormesh/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"tensormesh/MiniMaxAI/MiniMax-M2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":196608,"max_output_tokens":196608,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":196608}}]},"tensormesh/google/gemma-4-31B-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_input_tokens":32768,"max_output_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"tensormesh/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"tensormesh/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_system_messages":true,"supports_reasoning":true,"source":"https://serverless.tensormesh.ai/v1/models/openrouter","provider":"tensormesh","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"deepseek/deepseek-flash":{"mode":"chat","base_model":"deepseek-flash","max_input_tokens":1048576,"max_output_tokens":393216,"max_tokens":393216,"off_peak_pricing":{"cache_read_input_token_cost":3e-9,"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"windows":[{"hours_utc":["00:00-01:00","04:00-06:00","10:00-00:00"],"weekdays":[1,2,3,4,5]},{"hours_utc":"00:00-00:00","weekdays":[6,7]}]},"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/responses","/v1/messages"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"deepseek","peak_hours":{"timezone":"UTC","windows":[{"days":[1,2,3,4,5],"start":"01:00","end":"04:00"},{"days":[1,2,3,4,5],"start":"06:00","end":"10:00"}]},"supported_modalities":["text","image"],"supported_output_modalities":["text"],"metadata":{"notes":"DeepSeek-V4.1-Flash. Legacy names deepseek-v4-flash and deepseek-v4-flash-vision-exp route here at Flash pricing; deepseek-v4-pro also routes here from 2026-09-14 until V4.1 Pro ships.","calculation":"Peak: $0.30/M cache-miss input, $0.006/M cache-hit input, $1.20/M output. Off-peak = 0.5x."},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"deepseek/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"off_peak_pricing":{"cache_read_input_token_cost":3e-9,"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"windows":[{"hours_utc":["00:00-01:00","04:00-06:00","10:00-00:00"],"weekdays":[1,2,3,4,5]},{"hours_utc":"00:00-00:00","weekdays":[6,7]}]},"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"deepseek","peak_hours":{"timezone":"UTC","windows":[{"days":[1,2,3,4,5],"start":"01:00","end":"04:00"},{"days":[1,2,3,4,5],"start":"06:00","end":"10:00"}]},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"deepseek/deepseek-v4-flash-vision-exp":{"mode":"chat","base_model":"deepseek-v4-flash-vision-exp","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"off_peak_pricing":{"cache_read_input_token_cost":3e-9,"input_cost_per_token":1.5e-7,"output_cost_per_token":6e-7,"windows":[{"hours_utc":["00:00-01:00","04:00-06:00","10:00-00:00"],"weekdays":[1,2,3,4,5]},{"hours_utc":"00:00-00:00","weekdays":[6,7]}]},"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"deepseek","peak_hours":{"timezone":"UTC","windows":[{"days":[1,2,3,4,5],"start":"01:00","end":"04:00"},{"days":[1,2,3,4,5],"start":"06:00","end":"10:00"}]},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"deepseek/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"off_peak_pricing":{"cache_read_input_token_cost":2.2e-8,"input_cost_per_token":6.6e-7,"output_cost_per_token":0.00000198,"windows":[{"hours_utc":["00:00-01:00","04:00-06:00","10:00-00:00"],"weekdays":[1,2,3,4,5]},{"hours_utc":"00:00-00:00","weekdays":[6,7]}]},"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"deepseek","peak_hours":{"timezone":"UTC","windows":[{"days":[1,2,3,4,5],"start":"01:00","end":"04:00"},{"days":[1,2,3,4,5],"start":"06:00","end":"10:00"}]},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"tencent/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://www.tencentcloud.com/products/tokenhub","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"provider":"tencent","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"tencent/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://www.tencentcloud.com/products/tokenhub","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"provider":"tencent","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"tencent/minimax-m3":{"mode":"chat","base_model":"minimax-m3","max_input_tokens":1000000,"source":"https://www.tencentcloud.com/products/tokenhub","supported_endpoints":["/v1/chat/completions"],"supports_adaptive_thinking":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":false,"provider":"tencent","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"cognition/swe-1.6":{"mode":"chat","base_model":"swe-1.6","supports_function_calling":true,"supports_prompt_caching":true,"source":"https://docs.devin.ai/windsurf/plugins/cascade/models","provider":"cognition","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"cognition/swe-1.7":{"mode":"chat","base_model":"swe-1.7","supports_function_calling":true,"supports_prompt_caching":true,"source":"https://docs.devin.ai/desktop/models","provider":"cognition","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"cognition/swe-1.7-lightning":{"mode":"chat","base_model":"swe-1.7-lightning","supports_function_calling":true,"supports_prompt_caching":true,"source":"https://docs.devin.ai/desktop/models","provider":"cognition","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"pinstripes/ps/glm-4.5-air":{"mode":"chat","base_model":"glm-4.5-air","max_tokens":128000,"max_input_tokens":128000,"max_output_tokens":128000,"supports_function_calling":true,"supports_assistant_prefill":true,"supports_reasoning":true,"source":"https://pinstripes.io/","provider":"pinstripes","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"pinstripes/ps/qwen3.6-35b-a3b":{"mode":"chat","base_model":"qwen3.6-35b-a3b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_assistant_prefill":true,"supports_reasoning":true,"source":"https://pinstripes.io/","provider":"pinstripes","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"pinstripes/ps/qwen3-30b-a3b":{"mode":"chat","base_model":"qwen3-30b-a3b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_assistant_prefill":true,"supports_reasoning":true,"source":"https://pinstripes.io/","provider":"pinstripes","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"pinstripes/ps/qwen3-coder-30b-a3b":{"mode":"chat","base_model":"qwen3-coder-30b-a3b","max_tokens":131072,"max_input_tokens":131072,"max_output_tokens":131072,"supports_function_calling":true,"supports_assistant_prefill":true,"supports_reasoning":false,"source":"https://pinstripes.io/","provider":"pinstripes","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"pinstripes/ps/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_tokens":163840,"max_input_tokens":163840,"max_output_tokens":163840,"supports_function_calling":true,"supports_assistant_prefill":true,"supports_reasoning":true,"source":"https://pinstripes.io/","provider":"pinstripes","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"pinstripes/ps/minimax-m2.7":{"mode":"chat","base_model":"minimax-m2.7","max_tokens":1000192,"max_input_tokens":1000192,"max_output_tokens":1000192,"supports_function_calling":true,"supports_assistant_prefill":true,"supports_reasoning":false,"source":"https://pinstripes.io/","provider":"pinstripes","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000192}}]},"darkbloom/gemma-4-26b":{"mode":"chat","base_model":"gemma-4-26b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.darkbloom.dev/","supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_native_streaming":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"darkbloom","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"darkbloom/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.darkbloom.dev/","supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_native_streaming":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"darkbloom","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"xai/grok-4.20-0309-non-reasoning":{"mode":"chat","base_model":"grok-4.20-0309-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-multi-agent-0309":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"supported_endpoints":["/v1/responses"],"provider":"xai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-build-0.1":{"mode":"chat","base_model":"grok-build-0.1","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://docs.x.ai/developers/models/grok-build-0.1","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"provider":"xai","supported_modalities":["text","image"],"supported_output_modalities":["text"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"claude-mythos-5":{"mode":"chat","base_model":"claude-mythos-5","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"source":"https://platform.claude.com/docs/en/about-claude/pricing","supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"provider_specific_entry":{"us":1.1},"provider":"anthropic","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"claude-mythos-5-1":{"mode":"chat","base_model":"claude-mythos-5-1","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"provider_specific_entry":{"us":1.1},"supports_output_config":true,"prompt_cache_min_tokens":512,"supports_native_structured_output":true,"source":"https://platform.claude.com/docs/en/about-claude/pricing","provider":"anthropic","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"claude-mythos-preview":{"mode":"chat","base_model":"claude-mythos","supports_anthropic_compaction":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"source":"https://platform.claude.com/docs/en/about-claude/models/overview","supports_adaptive_thinking":true,"thinking_always_on":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"provider_specific_entry":{"us":1.1},"provider":"anthropic","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"gemini/gemini-robotics-er-2-streaming-preview":{"mode":"chat","base_model":"gemini-robotics-er-2-streaming","source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text"],"supports_audio_input":true,"supports_function_calling":true,"supports_video_input":true,"supports_vision":true,"supports_web_search":true,"web_search_billing_unit":"per_query","provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"mistral/mistral-small-2603":{"mode":"chat","base_model":"mistral-small","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-small-4-0-26-03","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/labs-leanstral-1-5":{"mode":"chat","base_model":"labs-leanstral-1-5","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.mistral.ai/models/model-cards/leanstral-1-5","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/models/deepseek-v4p1-flash":{"mode":"chat","base_model":"deepseek-flash","max_input_tokens":1048576,"max_output_tokens":393216,"max_tokens":393216,"source":"https://fireworks.ai/models/fireworks/deepseek-v4p1-flash","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supported_endpoints":["/v1/chat/completions"],"supports_service_tier":true,"supports_system_messages":true,"supports_native_streaming":true,"metadata":{"notes":"DeepSeek-V4.1-Flash on Fireworks serverless (552B MoE, 1M context, image input). Priority serving tier is 1.25x Standard across all token types.","calculation":"Standard: $0.22/M uncached input, $0.007/M cached input, $0.66/M output. Priority: $0.275/M, $0.00875/M, $0.825/M."},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-vision-exp":{"mode":"chat","base_model":"deepseek-v4-flash-vision","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_response_schema":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"fireworks_ai/accounts/fireworks/models/kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/deepseek-v4p1-flash":{"mode":"chat","base_model":"deepseek-v4.1-flash","max_input_tokens":1048576,"max_output_tokens":393216,"max_tokens":393216,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"fireworks_ai/deepseek-v4-flash-vision-exp":{"mode":"chat","base_model":"deepseek-v4-flash-vision","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_response_schema":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"fireworks_ai/glm-5p2-fast":{"mode":"chat","base_model":"glm-5.2-fast","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/glm-5p2-fast-us":{"mode":"chat","base_model":"glm-5.2-fast-us","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/kimi-k3-fast":{"mode":"chat","base_model":"kimi-k3-fast","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/kimi-k3-us":{"mode":"chat","base_model":"kimi-k3-us","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/qwen3p8-max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":262144,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"fireworks_ai/muse-glimmer-30b":{"mode":"chat","base_model":"muse-glimmer-30b","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"fireworks_ai/nemotron-lightning-3p5-30b-a3b":{"mode":"chat","base_model":"nemotron-lightning-3.5-30b-a3b","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/nemotron-3-ultra-nvfp4":{"mode":"chat","base_model":"nemotron-3-ultra-nvfp4","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/accounts/fireworks/models/muse-glimmer-30b":{"mode":"chat","base_model":"muse-glimmer-30b","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"fireworks_ai/accounts/fireworks/models/nemotron-lightning-3p5-30b-a3b":{"mode":"chat","base_model":"nemotron-lightning-3.5-30b-a3b","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/accounts/fireworks/models/nemotron-3-ultra-nvfp4":{"mode":"chat","base_model":"nemotron-3-ultra-nvfp4","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"fireworks_ai/accounts/fireworks/models/qwen3p8-max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":262144,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"fireworks_ai/accounts/fireworks/routers/glm-5p2-fast":{"mode":"chat","base_model":"accounts/fireworks/routers/glm-5.2-fast","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/routers/glm-5p2-fast-us":{"mode":"chat","base_model":"accounts/fireworks/routers/glm-5.2-fast-us","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/routers/kimi-k3-fast":{"mode":"chat","base_model":"accounts/fireworks/routers/kimi-k3-fast","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/accounts/fireworks/routers/kimi-k3-us":{"mode":"chat","base_model":"accounts/fireworks/routers/kimi-k3-us","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"reasoning_effort_levels":["low","high","max"],"source":"https://docs.fireworks.ai/serverless/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/zai-org/glm-5.3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/deepseek/deepseek-v4-pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1048576,"max_output_tokens":393216,"max_tokens":393216,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"novita/moonshotai/kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"novita/tencent/hy3":{"mode":"chat","base_model":"hy3","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"novita/zai-org/glm-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/moonshotai/kimi-k2.7-code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"novita/deepseek/deepseek-v4-flash-vision-exp":{"mode":"chat","base_model":"deepseek-v4-flash-vision","max_input_tokens":1048576,"max_output_tokens":393216,"max_tokens":393216,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"novita/deepseek/deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1048576,"max_output_tokens":393216,"max_tokens":393216,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"novita/mindai/macaron-v1-venti":{"mode":"chat","base_model":"mindai/macaron-v1-venti","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/minimax/minimax-m3":{"mode":"chat","base_model":"minimax-m3","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/deepseek/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1048576,"max_output_tokens":393216,"max_tokens":393216,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"novita/deepseek/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1048576,"max_output_tokens":393216,"max_tokens":393216,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"novita/inclusionai/ling-3.0-flash-fast":{"mode":"chat","base_model":"inclusionai/ling-3.0-flash-fast","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"novita/qwen/qwen3.8-max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/inclusionai/ling-3.0-flash":{"mode":"chat","base_model":"inclusionai/ling-3.0-flash","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"novita/mindai/macaron-v1-tall":{"mode":"chat","base_model":"mindai/macaron-v1-tall","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"novita/stepfun/step-3.7-flash":{"mode":"chat","base_model":"stepfun/step-3.7-flash","max_input_tokens":262144,"max_output_tokens":256000,"max_tokens":256000,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"novita/nvidia/nemotron-3-nano-30b-a3b":{"mode":"chat","base_model":"nemotron-3-nano-30b-a3b","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"novita/baidu/cobuddy":{"mode":"chat","base_model":"cobuddy","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/xiaomimimo/mimo-v2.5":{"mode":"chat","base_model":"mimo","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/qwen/qwen3.7-max":{"mode":"chat","base_model":"qwen3.7-max","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/xiaomimimo/mimo-v2.5-pro":{"mode":"chat","base_model":"mimo-v2.5-pro","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/qwen/qwen3.6-27b":{"mode":"chat","base_model":"qwen3.6-27b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/moonshotai/kimi-k2.6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"novita/zai-org/glm-5.1":{"mode":"chat","base_model":"glm-5.1","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/minimax/minimax-m2.7-highspeed":{"mode":"chat","base_model":"minimax-m2.7-highspeed","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/zai-org/glm-5v-turbo":{"mode":"chat","base_model":"glm-5v-turbo","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/google/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"gemma-4-26b-a4b-it","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/google/gemma-4-31b-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/zai-org/glm-5-turbo":{"mode":"chat","base_model":"glm-5-turbo","max_input_tokens":202800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/minimax/minimax-m2.7":{"mode":"chat","base_model":"minimax-m2.7","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/minimax/minimax-m2.5-highspeed":{"mode":"chat","base_model":"minimax-m2.5-highspeed","max_input_tokens":204800,"max_output_tokens":131100,"max_tokens":131100,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131100}}]},"novita/qwen/qwen3.5-27b":{"mode":"chat","base_model":"qwen3.5-27b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/qwen/qwen3.5-122b-a10b":{"mode":"chat","base_model":"qwen3.5-122b-a10b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/qwen/qwen3.5-35b-a3b":{"mode":"chat","base_model":"qwen3.5-35b-a3b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/qwen/qwen3.5-397b-a17b":{"mode":"chat","base_model":"qwen3.5-397b-a17b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/minimax/minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","max_input_tokens":204800,"max_output_tokens":131100,"max_tokens":131100,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131100}}]},"novita/zai-org/glm-5":{"mode":"chat","base_model":"glm-5","max_input_tokens":202800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/qwen/qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/deepseek/deepseek-ocr-2":{"mode":"chat","base_model":"deepseek-ocr-2","max_input_tokens":8192,"max_output_tokens":8192,"max_tokens":8192,"source":"https://novita.ai/pricing","supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"novita/moonshotai/kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"novita/zai-org/glm-4.7-h":{"mode":"chat","base_model":"glm-4.7-h","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"novita/zai-org/glm-4.7-flash":{"mode":"chat","base_model":"glm-4.7-flash","max_input_tokens":200000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"novita/qwen/qwen3.6-35b-a3b":{"mode":"chat","base_model":"qwen3.6-35b-a3b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://novita.ai/pricing","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"novita/deepseek/deepseek_v3":{"mode":"chat","base_model":"deepseek-v3","max_input_tokens":64000,"max_output_tokens":16000,"max_tokens":16000,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16000}}]},"novita/deepseek/deepseek-r1":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":64000,"max_output_tokens":16000,"max_tokens":16000,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16000}}]},"novita/deepseek/deepseek-v3/community":{"mode":"chat","base_model":"deepseek-v3/community","max_input_tokens":64000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"novita/deepseek/deepseek-r1/community":{"mode":"chat","base_model":"deepseek-r1/community","max_input_tokens":64000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"novita/thudm/glm-4-32b-0414":{"mode":"chat","base_model":"thudm/glm-4-32b","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://api.novita.ai/v3/openai/models","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"novita/meta-llama/llama-3.2-1b-instruct":{"mode":"chat","base_model":"llama-3.2-1b-instruct","max_input_tokens":131000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://api.novita.ai/v3/openai/models","supports_response_schema":true,"supports_vision":false,"provider":"novita","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"wandb/deepseek-ai/DeepSeek-V4-Flash":{"mode":"chat","base_model":"deepseek-v4-flash","supports_reasoning":true,"max_tokens":1048576,"max_input_tokens":1049000,"deprecation_date":"2026-10-05","supports_prompt_caching":true,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wandb/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_prompt_caching":true,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/deepseek-ai/DeepSeek-V4-Pro":{"mode":"chat","base_model":"deepseek-v4-pro","supports_reasoning":true,"max_tokens":1048576,"max_input_tokens":1049000,"deprecation_date":"2026-10-05","supports_prompt_caching":true,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wandb/google/gemma-4-31B-it":{"mode":"chat","base_model":"gemma-4-31b-it","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_vision":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/ibm-granite/granite-4.1-8b":{"mode":"chat","base_model":"granite-4.1-8b","deprecation_date":"2026-10-05","max_tokens":131072,"max_input_tokens":131000,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"wandb/JetBrains/Mellum2-12B-A2.5B-Instruct":{"mode":"chat","base_model":"jetbrains/mellum2-12b-a2.5b-instruct","deprecation_date":"2026-10-05","max_tokens":131072,"max_input_tokens":131000,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"wandb/meta-llama/Llama-3.1-70B-Instruct":{"mode":"chat","base_model":"llama-3.1-70b-instruct","deprecation_date":"2026-10-05","max_tokens":128000,"max_input_tokens":131000,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"wandb/MiniMaxAI/MiniMax-M3":{"mode":"chat","base_model":"minimax-m3","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_prompt_caching":true,"supports_vision":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/moonshotai/Kimi-K2.7-Code":{"mode":"chat","base_model":"kimi-k2.7-code","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_prompt_caching":true,"supports_vision":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"kimi-k2.6","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_prompt_caching":true,"supports_vision":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B":{"mode":"chat","base_model":"nvidia-nemotron-3.5-lightning-30b-a3b","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_prompt_caching":true,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"mode":"chat","base_model":"nvidia-nemotron-3-ultra-550b-a55b","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_prompt_caching":true,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/OpenPipe/Qwen3-14B-Instruct":{"mode":"chat","base_model":"openpipe/qwen3-14b-instruct","deprecation_date":"2026-10-05","max_tokens":32768,"max_input_tokens":32800,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"wandb/Qwen/Qwen3.8-27B":{"mode":"chat","base_model":"qwen3.8-27b","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_prompt_caching":true,"supports_vision":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/Qwen/Qwen3.6-35B-A3B":{"mode":"chat","base_model":"qwen3.6-35b-a3b","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_vision":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/Qwen/Qwen3.6-27B":{"mode":"chat","base_model":"qwen3.6-27b","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"deprecation_date":"2026-10-05","supports_prompt_caching":true,"supports_vision":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/Qwen/Qwen3.5-35B-A3B":{"mode":"chat","base_model":"qwen3.5-35b-a3b","deprecation_date":"2026-10-05","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":262000,"supports_vision":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/Qwen/Qwen3-30B-A3B-Instruct-2507":{"mode":"chat","base_model":"qwen3-30b-a3b-instruct","deprecation_date":"2026-10-05","max_tokens":262144,"max_input_tokens":262000,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wandb/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","supports_reasoning":true,"max_input_tokens":1049000,"supports_prompt_caching":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"wandb/ibm-granite/granite-4.2-8b":{"mode":"chat","base_model":"granite-4.2-8b","supports_reasoning":true,"max_input_tokens":131000,"supports_prompt_caching":true,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"wandb/zai-org/GLM-5.2":{"mode":"chat","base_model":"glm-5.2","supports_reasoning":true,"max_tokens":262144,"max_input_tokens":1049000,"supports_prompt_caching":true,"supports_vision":false,"source":"https://wandb.ai/site/pricing/tokens/","provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/openai/gpt-oss-120b-Turbo":{"mode":"chat","base_model":"gpt-oss-120b-turbo","max_tokens":131072,"max_input_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"deepinfra/MiniMaxAI/MiniMax-M2.7":{"mode":"chat","base_model":"minimax-m2.7","max_tokens":196608,"max_input_tokens":196608,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":196608}}]},"deepinfra/Qwen/Qwen3.8-27B":{"mode":"chat","base_model":"qwen3.8-27b","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/google/gemma-4-31B-it-Ultra":{"mode":"chat","base_model":"gemma-4-31b-it-ultra","max_tokens":131072,"max_input_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"deepinfra/moonshotai/Kimi-K2.5":{"mode":"chat","base_model":"kimi-k2.5","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/zai-org/GLM-4.7-Flash":{"mode":"chat","base_model":"glm-4.7-flash","max_tokens":202752,"max_input_tokens":202752,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"deepinfra/zai-org/GLM-4.6":{"mode":"chat","base_model":"glm-4.6","max_tokens":202752,"max_input_tokens":202752,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"deepinfra/anthropic/claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","max_input_tokens":1000000,"max_tokens":1000000,"prompt_cache_min_tokens":1024,"source":"https://deepinfra.com/pricing","supports_adaptive_thinking":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/anthropic/claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","max_tokens":1000000,"max_input_tokens":1000000,"prompt_cache_min_tokens":1024,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_adaptive_thinking":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/google/gemini-3.5-flash":{"mode":"chat","base_model":"gemini-3.5-flash","max_tokens":1000000,"max_input_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","supports_audio_input":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/XiaomiMiMo/MiMo-V2.5":{"mode":"chat","base_model":"mimo-v2.5","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/Qwen/Qwen3-Max":{"mode":"chat","base_model":"qwen3-max","max_tokens":256000,"max_input_tokens":256000,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"deepinfra/google/gemma-4-31B-it-turbo":{"mode":"chat","base_model":"gemma-4-31b-it-turbo","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/thinkingmachines/Inkling-Small":{"mode":"chat","base_model":"thinkingmachines/inkling-small","max_tokens":524288,"max_input_tokens":524288,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"deepinfra/meta-models/Muse-Glimmer-30B":{"mode":"chat","base_model":"models/muse-glimmer-30b","max_tokens":131072,"max_input_tokens":131072,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"deepinfra/Qwen/Qwen3-Max-Thinking":{"mode":"chat","base_model":"qwen3-max-thinking","max_tokens":256000,"max_input_tokens":256000,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"deepinfra/Qwen/Qwen3-VL-235B-A22B-Instruct":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/Qwen/Qwen3-VL-30B-A3B-Instruct":{"mode":"chat","base_model":"qwen3-vl-30b-a3b-instruct","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/Qwen/Qwen3.5-27B":{"mode":"chat","base_model":"qwen3.5-27b","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/Qwen/Qwen3.6-35B-A3B":{"mode":"chat","base_model":"qwen3.6-35b-a3b","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/nvidia/Nemotron-Content-Safety-3.5":{"mode":"chat","base_model":"nemotron-content-safety-3.5","max_tokens":131072,"max_input_tokens":131072,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"deepinfra/anthropic/claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","max_input_tokens":1000000,"max_tokens":1000000,"prompt_cache_min_tokens":512,"source":"https://deepinfra.com/pricing","supports_adaptive_thinking":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/thinkingmachines/Inkling":{"mode":"chat","base_model":"thinkingmachines/inkling","max_tokens":524288,"max_input_tokens":524288,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"deepinfra/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"kimi-k2.6","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","max_tokens":1048576,"max_input_tokens":1048576,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"deepinfra/Qwen/Qwen3.7-Max":{"mode":"chat","base_model":"qwen3.7-max","max_tokens":256000,"max_input_tokens":256000,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"deepinfra/ByteDance/Seed-2.0-mini":{"mode":"chat","base_model":"seed-2.0-mini","max_tokens":256000,"max_input_tokens":256000,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"deepinfra/Qwen/Qwen3.8-2.4T-A95B":{"mode":"chat","base_model":"qwen3.8-2.4t-a95b","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/MiniMaxAI/MiniMax-M3":{"mode":"chat","base_model":"minimax-m3","max_tokens":524288,"max_input_tokens":524288,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"deepinfra/google/gemini-3.1-flash-lite":{"mode":"chat","base_model":"gemini-3.1-flash-lite","max_tokens":1000000,"max_input_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","supports_audio_input":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/google/gemini-3.7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","max_tokens":1000000,"max_input_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","supports_audio_input":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/inclusionAI/Ling-3.0-flash":{"mode":"chat","base_model":"inclusionai/ling-3.0-flash","max_tokens":131072,"max_input_tokens":131072,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"deepinfra/stepfun-ai/Step-3.7-Flash":{"mode":"chat","base_model":"stepfun-ai/step-3.7-flash","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/Qwen/Qwen3.5-35B-A3B":{"mode":"chat","base_model":"qwen3.5-35b-a3b","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/ByteDance/Seed-1.8":{"mode":"chat","base_model":"seed-1.8","max_tokens":256000,"max_input_tokens":256000,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"deepinfra/tencent/Hy3":{"mode":"chat","base_model":"hy3","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/ByteDance/Seed-2.0-code":{"mode":"chat","base_model":"seed-2.0-code","max_tokens":256000,"max_input_tokens":256000,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"deepinfra/ByteDance/Seed-2.0-pro":{"mode":"chat","base_model":"seed-2.0-pro","max_tokens":256000,"max_input_tokens":256000,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"deepinfra/zai-org/GLM-5":{"mode":"chat","base_model":"glm-5","max_tokens":202752,"max_input_tokens":202752,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B":{"mode":"chat","base_model":"nemotron-3-nano-30b-a3b","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/moonshotai/Kimi-K2.7-Code":{"mode":"chat","base_model":"kimi-k2.7-code","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/anthropic/claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","max_input_tokens":1000000,"max_tokens":1000000,"prompt_cache_min_tokens":1024,"source":"https://deepinfra.com/pricing","supports_adaptive_thinking":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/Qwen/Qwen3.5-397B-A17B":{"mode":"chat","base_model":"qwen3.5-397b-a17b","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_tokens":1048576,"max_input_tokens":1048576,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"deepinfra/google/gemma-4-E4B-it":{"mode":"chat","base_model":"gemma-4-e4b-it","max_tokens":131072,"max_input_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"deepinfra/deepseek-ai/DeepSeek-V3.2":{"mode":"chat","base_model":"deepseek-v3.2","max_tokens":163840,"max_input_tokens":163840,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"deepinfra/Qwen/Qwen3.8-Max":{"mode":"chat","base_model":"qwen3.8-max","max_tokens":256000,"max_input_tokens":256000,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"deepinfra/anthropic/claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","max_input_tokens":1000000,"max_tokens":1000000,"prompt_cache_min_tokens":512,"source":"https://deepinfra.com/pricing","supports_adaptive_thinking":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"thinking_always_on":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"mode":"chat","base_model":"nvidia-nemotron-3-ultra-550b-a55b","max_tokens":262144,"max_input_tokens":262144,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/Qwen/Qwen3.5-122B-A10B":{"mode":"chat","base_model":"qwen3.5-122b-a10b","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/zai-org/GLM-5.1":{"mode":"chat","base_model":"glm-5.1","max_tokens":202752,"max_input_tokens":202752,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"deepinfra/deepseek-ai/DeepSeek-V4-Pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_tokens":1048576,"max_input_tokens":1048576,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"deepinfra/nvidia/NVIDIA-Nemotron-3-Super-120B-A12B":{"mode":"chat","base_model":"nvidia-nemotron-3-super-120b-a12b","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/zai-org/GLM-5.2":{"mode":"chat","base_model":"glm-5.2","max_tokens":1048576,"max_input_tokens":1048576,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"deepinfra/moonshotai/Kimi-K3":{"mode":"chat","base_model":"kimi-k3","max_tokens":1048576,"max_input_tokens":1048576,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"deepinfra/anthropic/claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","max_input_tokens":1000000,"max_tokens":1000000,"prompt_cache_min_tokens":2048,"source":"https://deepinfra.com/pricing","supports_adaptive_thinking":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/Qwen/Qwen3.6-27B":{"mode":"chat","base_model":"qwen3.6-27b","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/google/gemma-4-26B-A4B-it":{"mode":"chat","base_model":"gemma-4-26b-a4b-it","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/google/gemini-3.1-pro":{"mode":"chat","base_model":"gemini-3.1-pro","max_tokens":1000000,"max_input_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","supports_audio_input":true,"provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"deepinfra/XiaomiMiMo/MiMo-V2.5-Pro":{"mode":"chat","base_model":"mimo-v2.5-pro","max_tokens":1048576,"max_input_tokens":1048576,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"deepinfra/anthropic/claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","max_tokens":200000,"max_input_tokens":200000,"prompt_cache_min_tokens":4096,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":200000}}]},"deepinfra/deepseek-ai/DeepSeek-V4-Flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_tokens":1048576,"max_input_tokens":1048576,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"deepinfra/openai/gpt-oss-120b-Ultra":{"mode":"chat","base_model":"gpt-oss-120b-ultra","max_tokens":131072,"max_input_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"deepinfra/Qwen/Qwen3.5-9B":{"mode":"chat","base_model":"qwen3.5-9b","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepinfra/MiniMaxAI/MiniMax-M2.7-Turbo":{"mode":"chat","base_model":"minimax-m2.7-turbo","max_tokens":196608,"max_input_tokens":196608,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":196608}}]},"deepinfra/zai-org/GLM-4.7":{"mode":"chat","base_model":"glm-4.7","max_tokens":202752,"max_input_tokens":202752,"supports_prompt_caching":true,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":false,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"deepinfra/google/gemma-4-31B-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_tokens":262144,"max_input_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_reasoning":true,"supports_vision":true,"source":"https://deepinfra.com/pricing","provider":"deepinfra","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"gemini/gemini-omni-1.1-flash":{"mode":"chat","base_model":"gemini-omni-1.1-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1beta/interactions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","video"],"supports_audio_input":true,"supports_reasoning":true,"supports_system_messages":true,"supports_video_input":true,"supports_vision":true,"tpm":800000,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"xai/grok-4.20":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-reasoning-latest":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-non-reasoning":{"mode":"chat","base_model":"grok-4.20-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-non-reasoning-latest":{"mode":"chat","base_model":"grok-4.20-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-multi-agent":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supported_endpoints":["/v1/responses"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-multi-agent-latest":{"mode":"responses","base_model":"grok-4.20-multi-agent","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supported_endpoints":["/v1/responses"],"supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"groq/qwen/qwen3.8-27b":{"mode":"chat","base_model":"qwen3.8-27b","max_input_tokens":131042,"max_output_tokens":16384,"max_tokens":16384,"source":"https://console.groq.com/docs/model/qwen/qwen3.8-27b","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"groq","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"mistral/mistral-medium-3.5":{"mode":"chat","base_model":"mistral-medium-3.5","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-medium-3-5-26-04","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/mistral-vibe-cli-latest":{"mode":"chat","base_model":"mistral-vibe-cli","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-medium-3-5-26-04","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/mistral-vibe-cli-with-tools":{"mode":"chat","base_model":"mistral-vibe-cli-with-tools","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-medium-3-5-26-04","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/mistral-vibe-cli-fast":{"mode":"chat","base_model":"mistral-vibe-cli-fast","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"reasoning_effort_levels":["none","high"],"source":"https://docs.mistral.ai/models/model-cards/mistral-small-4-0-26-03","supports_assistant_prefill":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"mistral/mistral-code-latest":{"mode":"chat","base_model":"mistral-code","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://docs.mistral.ai/models/model-cards/codestral-25-08","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"mistral/mistral-code-fim-latest":{"mode":"chat","base_model":"mistral-code-fim","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://docs.mistral.ai/models/model-cards/codestral-25-08","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"mistral/mistral-code-agent-latest":{"mode":"chat","base_model":"mistral-code-agent","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://mistral.ai/news/devstral-2-vibe-cli","supports_assistant_prefill":true,"supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"mistral/labs-leanstral-1-5-1":{"mode":"chat","base_model":"labs-leanstral-1-5-1","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"source":"https://docs.mistral.ai/models/model-cards/leanstral-1-5","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"mistral","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"fireworks_ai/glm-5p3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048576,"max_output_tokens":128000,"max_tokens":128000,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"fireworks_ai/glm-5p3-flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://api.fireworks.ai/v1/serverless/models","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"fireworks_ai/accounts/fireworks/models/inkling":{"mode":"chat","base_model":"inkling","max_input_tokens":1048576,"max_tokens":1048576,"source":"https://fireworks.ai/models/fireworks/inkling","supports_function_calling":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"fireworks_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"zai/glm-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1000000,"max_output_tokens":128000,"source":"https://docs.z.ai/guides/overview/pricing","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"provider":"zai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"together_ai/Qwen/Qwen3.8-Flash":{"mode":"chat","base_model":"qwen3.8-flash","max_input_tokens":1000000,"max_tokens":1000000,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"together_ai/moonshotai/Kimi-K2.5-fp4":{"mode":"chat","base_model":"kimi-k2.5-fp4","max_input_tokens":262144,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/MiniMaxAI/MiniMax-M2.7":{"mode":"chat","base_model":"minimax-m2.7","max_input_tokens":196608,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/deepseek-ai/DeepSeek-R1-0528":{"mode":"chat","base_model":"deepseek-r1","max_input_tokens":163840,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/mistralai/Ministral-3-14B-Instruct-2512":{"mode":"chat","base_model":"ministral-3-14b-instruct","max_input_tokens":262144,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/nvidia/NVIDIA-Nemotron-Nano-9B-v2":{"mode":"chat","base_model":"nvidia-nemotron-nano-9b-v2","max_input_tokens":131072,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/mistralai/Mistral-7B-Instruct-v0.3":{"mode":"chat","base_model":"mistral-7b-instruct","max_input_tokens":32768,"source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"cerebras/gemma-4-31b":{"mode":"chat","base_model":"gemma-4-31b","max_input_tokens":131072,"max_output_tokens":40960,"max_tokens":40960,"source":"https://api.cerebras.ai/public/v1/models/gemma-4-31b","supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"cerebras","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"scaleway/glm-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":256000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://www.scaleway.com/en/pricing/model-as-a-service/","supports_function_calling":true,"supports_reasoning":true,"supports_vision":false,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"scaleway/deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":256000,"max_output_tokens":32768,"max_tokens":32768,"source":"https://www.scaleway.com/en/pricing/model-as-a-service/","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":false,"provider":"scaleway","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"azure/kimi-k2.7-code":{"mode":"chat","base_model":"kimi-k2.7-code","deprecation_date":"2026-10-03","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"bedrock/us-gov-west-1/nvidia.nemotron-nano-3-30b":{"mode":"chat","base_model":"nemotron-nano-3-30b","max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock/us-gov-west-1/nvidia.nemotron-nano-12b-v2":{"mode":"chat","base_model":"nemotron-nano-12b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supports_system_messages":true,"supports_vision":true,"supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock/us-gov-west-1/nvidia.nemotron-nano-9b-v2":{"mode":"chat","base_model":"nemotron-nano-9b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supports_system_messages":true,"supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock/us-gov-west-1/nvidia.nemotron-super-3-120b":{"mode":"chat","base_model":"nemotron-super-3-120b","max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock/us-gov-west-1/openai.gpt-oss-120b-1:0":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock/us-gov-west-1/anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"thinking_always_on":true,"supports_forced_tool_use":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/us-gov-west-1/anthropic.claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1","us-gov-east-1","us-gov-west-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/us-gov-east-1/nvidia.nemotron-nano-3-30b":{"mode":"chat","base_model":"nemotron-nano-3-30b","max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_native_structured_output":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock/us-gov-east-1/nvidia.nemotron-nano-12b-v2":{"mode":"chat","base_model":"nemotron-nano-12b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supports_system_messages":true,"supports_vision":true,"supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock/us-gov-east-1/nvidia.nemotron-nano-9b-v2":{"mode":"chat","base_model":"nemotron-nano-9b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supports_system_messages":true,"supports_audio_input":false,"supports_function_calling":true,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock/us-gov-east-1/nvidia.nemotron-super-3-120b":{"mode":"chat","base_model":"nemotron-super-3-120b","max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://aws.amazon.com/bedrock/pricing/","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_audio_input":false,"supports_response_schema":true,"supports_vision":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock/us-gov-east-1/openai.gpt-oss-20b-1:0":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock/us-gov-east-1/openai.gpt-oss-120b-1:0":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock/us-gov-east-1/anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"supports_adaptive_thinking":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_max_reasoning_effort":true,"supports_mid_conversation_system":true,"supports_native_structured_output":false,"supports_output_config":true,"supports_parallel_tool_use_config":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"thinking_always_on":true,"supports_forced_tool_use":false,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/us-gov-east-1/anthropic.claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1","us-gov-east-1","us-gov-west-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-5.6-terra":{"mode":"responses","base_model":"gpt-5.6-terra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","us-gov-west-1","us-gov-east-1","us-west-1","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-5.6-luna":{"mode":"responses","base_model":"gpt-5.6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","us-gov-west-1","us-gov-east-1","us-west-1","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-5.4":{"mode":"responses","base_model":"gpt-5.4","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","us-gov-west-1"],"source":"https://aws.amazon.com/bedrock/pricing/","supports_service_tier":true,"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/xai.grok-4.3":{"mode":"responses","base_model":"grok-4.3","use_openai_responses_path":true,"max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"supported_endpoints":["/v1/responses"],"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-west-2","us-east-1","us-east-2","us-gov-west-1"],"model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock_mantle/us-gov-west-1/xai.grok-4.6":{"mode":"chat","base_model":"grok-4.6","use_openai_responses_path":true,"max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-west-2","us-gov-east-1","us-east-1","us-east-2","us-west-1","us-gov-west-1","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"bedrock_mantle/us-gov-west-1/google.gemma-4-e2b":{"mode":"chat","base_model":"gemma-4-e2b","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":true,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/google.gemma-4-26b-a4b":{"mode":"chat","base_model":"gemma-4-26b-a4b","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"bedrock_mantle/us-gov-west-1/google.gemma-4-31b":{"mode":"chat","base_model":"gemma-4-31b","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"use_openai_responses_path":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":false,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"bedrock_mantle/us-gov-east-1/openai.gpt-5.4":{"mode":"responses","base_model":"gpt-5.4","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_minimal_reasoning_effort":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_xhigh_reasoning_effort":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","us-gov-west-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-east-1/xai.grok-4.6":{"mode":"chat","base_model":"grok-4.6","use_openai_responses_path":true,"max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-west-2","us-gov-east-1","us-east-1","us-east-2","us-west-1","us-gov-west-1","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"bedrock_mantle/us-gov-east-1/openai.gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"bedrock_mantle/us-gov-east-1/openai.gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"azure/us-gov/gpt-5.1":{"mode":"chat","base_model":"us-gov/gpt-5.1","default_reasoning_effort":"none","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"supported_endpoints":["/v1/chat/completions","/v1/batch","/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_native_streaming":true,"supports_none_reasoning_effort":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/us-gov/o3-mini":{"mode":"chat","base_model":"us-gov/o3-mini","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":100000}}]},"gemini/lyria-3.5-clip-preview":{"mode":"chat","base_model":"lyria-3.5-clip","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_input":false,"supports_audio_output":true,"supports_function_calling":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_vision":false,"supports_web_search":false,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"gemini/lyria-3.5-pro-preview":{"mode":"chat","base_model":"lyria-3.5-pro","max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_input":false,"supports_audio_output":true,"supports_function_calling":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_vision":false,"supports_web_search":false,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"gemini/lyria-3.5":{"mode":"chat","base_model":"lyria-3.5","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_input":false,"supports_audio_output":true,"supports_function_calling":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_vision":false,"supports_web_search":false,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"perplexity/anthropic/claude-fable-5":{"mode":"responses","base_model":"claude-fable-5","supports_adaptive_thinking":true,"thinking_always_on":true,"supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-opus-5":{"mode":"responses","base_model":"claude-opus-5","supports_adaptive_thinking":true,"prompt_cache_min_tokens":512,"supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-opus-4-8":{"mode":"responses","base_model":"claude-opus-4-8","supports_adaptive_thinking":true,"supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-sonnet-5":{"mode":"responses","base_model":"claude-sonnet-5","supports_adaptive_thinking":true,"supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/anthropic/claude-sonnet-4-6":{"mode":"responses","base_model":"claude-sonnet-4-6","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5.6-sol":{"mode":"responses","base_model":"gpt-5.6-sol","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5.6-terra":{"mode":"responses","base_model":"gpt-5.6-terra","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5.6-luna":{"mode":"responses","base_model":"gpt-5.6-luna","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5.5":{"mode":"responses","base_model":"gpt-5.5","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5.4":{"mode":"responses","base_model":"gpt-5.4","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5.4-mini":{"mode":"responses","base_model":"gpt-5.4-mini","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5.4-nano":{"mode":"responses","base_model":"gpt-5.4-nano","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/openai/gpt-5":{"mode":"responses","base_model":"gpt-5","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-3.1-pro-preview":{"mode":"responses","base_model":"gemini-3.1-pro","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-3.1-flash-lite":{"mode":"responses","base_model":"gemini-3.1-flash-lite","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-3.5-flash":{"mode":"responses","base_model":"gemini-3.5-flash","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-3.5-flash-lite":{"mode":"responses","base_model":"gemini-3.5-flash-lite","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-3.6-flash":{"mode":"responses","base_model":"gemini-3.6-flash","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/google/gemini-3.7-flash":{"mode":"responses","base_model":"gemini-3.7-flash","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/xai/grok-4.6":{"mode":"responses","base_model":"grok-4.6","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/xai/grok-4.5":{"mode":"responses","base_model":"grok-4.5","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/xai/grok-4.3":{"mode":"responses","base_model":"grok-4.3","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/xai/grok-4.20-reasoning":{"mode":"responses","base_model":"grok-4.20","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/xai/grok-4.20-non-reasoning":{"mode":"responses","base_model":"grok-4.20-non","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/xai/grok-4.20-multi-agent":{"mode":"responses","base_model":"grok-4.20-multi-agent","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/perplexity/glm-5.3":{"mode":"responses","base_model":"glm-5.3","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/perplexity/glm-5.3-flash":{"mode":"responses","base_model":"glm-5.3-flash","supports_web_search":true,"supports_function_calling":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/perplexity/nemotron-3.5-lightning-30b-a3b":{"mode":"responses","base_model":"nemotron-3.5-lightning-30b-a3b","supports_web_search":true,"supports_function_calling":true,"supports_reasoning":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"perplexity/perplexity/nemotron-3-ultra-550b-a55b":{"mode":"responses","base_model":"nemotron-3-ultra-550b-a55b","supports_web_search":true,"supports_function_calling":true,"supports_reasoning":true,"source":"https://docs.perplexity.ai/docs/agent-api/models","provider":"perplexity","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/anthropic/claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_sampling_params":false,"supports_adaptive_thinking":true,"thinking_always_on":true,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"prompt_cache_min_tokens":512,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-fable-5.1":{"mode":"chat","base_model":"claude-fable-5.1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_sampling_params":false,"supports_adaptive_thinking":true,"thinking_always_on":true,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"prompt_cache_min_tokens":512,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-4.8":{"mode":"chat","base_model":"claude-opus-4.8","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_sampling_params":false,"supports_adaptive_thinking":true,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_sampling_params":false,"supports_adaptive_thinking":true,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/google/gemini-3.5-flash":{"mode":"chat","base_model":"gemini-3.5-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.5-flash-lite":{"mode":"chat","base_model":"gemini-3.5-flash-lite","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.6-flash":{"mode":"chat","base_model":"gemini-3.6-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.8-flash":{"mode":"chat","base_model":"gemini-3.8-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supports_code_execution":true,"supports_parallel_function_calling":true,"supports_service_tier":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/openai/gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.3-codex":{"mode":"chat","base_model":"gpt-5.3-codex","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.4":{"mode":"chat","base_model":"gpt-5.4","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.4-mini":{"mode":"chat","base_model":"gpt-5.4-mini","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.4-nano":{"mode":"chat","base_model":"gpt-5.4-nano","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.5":{"mode":"chat","base_model":"gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-luna-pro":{"mode":"chat","base_model":"gpt-5.6-luna-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-terra-pro":{"mode":"chat","base_model":"gpt-5.6-terra-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/x-ai/grok-4.20":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":2000000,"max_output_tokens":1800000,"max_tokens":1800000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1800000}}]},"openrouter/x-ai/grok-4.20-multi-agent":{"mode":"chat","base_model":"grok-4.20-multi-agent","max_input_tokens":2000000,"max_output_tokens":1800000,"max_tokens":1800000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1800000}}]},"openrouter/x-ai/grok-4.3":{"mode":"chat","base_model":"grok-4.3","max_input_tokens":1000000,"max_output_tokens":900000,"max_tokens":900000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":900000}}]},"openrouter/x-ai/grok-4.5":{"mode":"chat","base_model":"grok-4.5","max_input_tokens":500000,"max_output_tokens":450000,"max_tokens":450000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":450000}}]},"openrouter/x-ai/grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":500000,"max_output_tokens":450000,"max_tokens":450000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":450000}}]},"openrouter/x-ai/grok-build-0.1":{"mode":"chat","base_model":"grok-build-0.1","max_input_tokens":256000,"max_output_tokens":230400,"max_tokens":230400,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":false,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":230400}}]},"baseten/zai-org/GLM-5.3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048576,"max_output_tokens":262144,"max_tokens":262144,"source":"https://inference.baseten.co/v1/models","supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"openrouter/minimax/minimax-m3":{"mode":"chat","base_model":"minimax-m3","max_input_tokens":524288,"max_output_tokens":512000,"max_tokens":512000,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_audio_input":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":512000}}]},"openrouter/qwen/qwen3.7-plus":{"mode":"chat","base_model":"qwen3.7-plus","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_audio_input":false,"supports_pdf_input":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/openai/gpt-6-astra":{"mode":"chat","base_model":"gpt-6-astra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/openai/gpt-6-astra","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","supports_prompt_cache_breakpoints":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-astra-pro":{"mode":"chat","base_model":"gpt-6-astra-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/openai/gpt-6-astra-pro","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","supports_prompt_cache_breakpoints":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/qwen/qwen3.8-flash":{"mode":"chat","base_model":"qwen3.8-flash","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/z-ai/glm-5.3-flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/deepseek/deepseek-v4-flash-vision-exp":{"mode":"chat","base_model":"deepseek-v4-flash-vision-exp","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/z-ai/glm-5.3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_parallel_function_calling":true,"supports_pdf_input":false,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/qwen/qwen3.8-27b":{"mode":"chat","base_model":"qwen3.8-27b","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/qwen/qwen3.8-2.4t-a95b":{"mode":"chat","base_model":"qwen3.8-2.4t-a95b","max_input_tokens":1000000,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"openrouter/nvidia/nemotron-3.5-lightning:free":{"mode":"chat","base_model":"nemotron-3.5-lightning","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.8-max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/qwen/qwen3.8-max","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/qwen/qwen3.8-max-0902":{"mode":"chat","base_model":"qwen3.8-max-0902","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/deepseek/deepseek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash-0731","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_parallel_function_calling":true,"supports_pdf_input":false,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/qwen/qwen3.7-flash":{"mode":"chat","base_model":"qwen3.7-flash","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/poolside/laguna-s-2.1":{"mode":"chat","base_model":"laguna-s-2.1","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/poolside/laguna-s-2.1:free":{"mode":"chat","base_model":"poolside/laguna-s-2.1","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/moonshotai/kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/poolside/laguna-xs-2.1":{"mode":"chat","base_model":"laguna-xs-2.1","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":false,"supports_prompt_caching":true,"supports_response_schema":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/poolside/laguna-xs-2.1:free":{"mode":"chat","base_model":"poolside/laguna-xs-2.1","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/google/gemini-3.1-flash-lite-image":{"mode":"chat","base_model":"gemini-3.1-flash-lite-image","max_input_tokens":65536,"max_output_tokens":58982,"max_tokens":58982,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":58982}}]},"openrouter/google/gemini-3.1-flash-image":{"mode":"chat","base_model":"gemini-3.1-flash-image","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/google/gemini-3-pro-image":{"mode":"chat","base_model":"gemini-3-pro-image","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/z-ai/glm-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_parallel_function_calling":true,"supports_pdf_input":false,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/z-ai/glm-5.2:free":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":32768,"max_output_tokens":29491,"max_tokens":29491,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":29491}}]},"openrouter/moonshotai/kimi-k2.7-code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_parallel_function_calling":true,"supports_pdf_input":false,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/nvidia/nemotron-3.5-content-safety":{"mode":"chat","base_model":"nemotron-3.5-content-safety","max_input_tokens":131072,"max_output_tokens":117964,"max_tokens":117964,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":117964}}]},"openrouter/nvidia/nemotron-3.5-content-safety:free":{"mode":"chat","base_model":"nemotron-3.5-content-safety","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"openrouter/nvidia/nemotron-3-ultra-550b-a55b":{"mode":"chat","base_model":"nemotron-3-ultra-550b-a55b","max_input_tokens":202800,"max_output_tokens":182520,"max_tokens":182520,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":182520}}]},"openrouter/nvidia/nemotron-3-ultra-550b-a55b:free":{"mode":"chat","base_model":"nemotron-3-ultra-550b-a55b","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/minimax/minimax-m3:free":{"mode":"chat","base_model":"minimax-m3","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/minimax/minimax-m3:free","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/qwen/qwen3.7-max":{"mode":"chat","base_model":"qwen3.7-max","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/mistralai/mistral-medium-3-5":{"mode":"chat","base_model":"mistral-medium-3-5","max_input_tokens":262144,"max_output_tokens":209715,"max_tokens":209715,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":209715}}]},"openrouter/nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free":{"mode":"chat","base_model":"nemotron-3-nano-omni-30b-a3b","max_input_tokens":256000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_vision":true,"supports_audio_input":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.5-plus-20260420":{"mode":"chat","base_model":"qwen3.5-plus-20260420","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.6-flash":{"mode":"chat","base_model":"qwen3.6-flash","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.6-35b-a3b":{"mode":"chat","base_model":"qwen3.6-35b-a3b","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/qwen/qwen3.6-max-preview":{"mode":"chat","base_model":"qwen3.6-max-preview","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3.6-27b":{"mode":"chat","base_model":"qwen3.6-27b","max_input_tokens":262144,"max_output_tokens":262140,"max_tokens":262140,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262140}}]},"openrouter/openai/gpt-5.5-pro":{"mode":"chat","base_model":"gpt-5.5-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-chat-latest":{"mode":"chat","base_model":"gpt-chat-latest","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":false,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/deepseek/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1024000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"openrouter/moonshotai/kimi-k2.6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/moonshotai/kimi-k2.5","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_parallel_function_calling":true,"supports_pdf_input":false,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","supports_video_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"openrouter/google/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"gemma-4-26b-a4b-it","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/google/gemma-4-26b-a4b-it:free":{"mode":"chat","base_model":"gemma-4-26b-a4b-it","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/google/gemma-4-31b-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_input_tokens":262144,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/google/gemma-4-31b-it:free":{"mode":"chat","base_model":"gemma-4-31b-it","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/z-ai/glm-5v-turbo":{"mode":"chat","base_model":"glm-5v-turbo","deprecation_date":"2098-12-31","max_input_tokens":202752,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/minimax/minimax-m2.7":{"mode":"chat","base_model":"minimax-m2.7","max_input_tokens":204800,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/minimax/minimax-m2.7:free":{"mode":"chat","base_model":"minimax-m2.7","max_input_tokens":196608,"max_output_tokens":176947,"max_tokens":176947,"source":"https://openrouter.ai/minimax/minimax-m2.7:free","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":176947}}]},"openrouter/mistralai/mistral-small-2603":{"mode":"chat","base_model":"mistral-small-2603","max_input_tokens":262144,"max_output_tokens":209715,"max_tokens":209715,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":209715}}]},"openrouter/z-ai/glm-5-turbo":{"mode":"chat","base_model":"glm-5-turbo","deprecation_date":"2098-12-31","max_input_tokens":202752,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/nvidia/nemotron-3-super-120b-a12b":{"mode":"chat","base_model":"nemotron-3-super-120b-a12b","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/nvidia/nemotron-3-super-120b-a12b:free":{"mode":"chat","base_model":"nemotron-3-super-120b-a12b","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/qwen/qwen3.5-9b":{"mode":"chat","base_model":"qwen3.5-9b","max_input_tokens":256000,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/openai/gpt-5.4-pro":{"mode":"chat","base_model":"gpt-5.4-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_web_search":true,"provider":"openrouter","supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/google/gemini-3.1-flash-image-preview":{"mode":"chat","base_model":"gemini-3.1-flash-image-preview","max_input_tokens":65536,"max_output_tokens":58982,"max_tokens":58982,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":58982}}]},"openrouter/google/gemini-3.1-pro-preview-customtools":{"mode":"chat","base_model":"gemini-3.1-pro-preview-customtools","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3-max-thinking":{"mode":"chat","base_model":"qwen3-max-thinking","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/minimax/minimax-m2-her":{"mode":"chat","base_model":"minimax-m2-her","max_input_tokens":65536,"max_output_tokens":2048,"max_tokens":2048,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_tool_choice":false,"supports_vision":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":2048,"range":{"min":1,"max":2048}}]},"openrouter/openai/gpt-audio":{"mode":"chat","base_model":"gpt-audio","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_audio_input":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","supports_audio_output":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/openai/gpt-audio-mini":{"mode":"chat","base_model":"gpt-audio-mini","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_audio_input":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","supports_audio_output":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/nvidia/nemotron-3-nano-30b-a3b":{"mode":"chat","base_model":"nemotron-3-nano-30b-a3b","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/z-ai/glm-4.6v":{"mode":"chat","base_model":"glm-4.6v","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/google/gemini-3-pro-image-preview":{"mode":"chat","base_model":"gemini-3-pro-image-preview","max_input_tokens":65536,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_tool_choice":false,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/openai/gpt-5.1-codex":{"mode":"chat","base_model":"gpt-5.1-codex","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.1-codex-mini":{"mode":"chat","base_model":"gpt-5.1-codex-mini","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/moonshotai/kimi-k2-thinking":{"mode":"chat","base_model":"kimi-k2-thinking","max_input_tokens":262144,"max_output_tokens":98304,"max_tokens":98304,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":98304}}]},"openrouter/mistralai/voxtral-small-24b-2507":{"mode":"chat","base_model":"voxtral-small-24b-2507","max_input_tokens":32768,"max_output_tokens":26214,"max_tokens":26214,"source":"https://openrouter.ai/api/v1/models","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_pdf_input":true,"supports_audio_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":26214}}]},"openrouter/openai/gpt-oss-safeguard-20b":{"mode":"chat","base_model":"gpt-oss-safeguard-20b","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3-vl-32b-instruct":{"mode":"chat","base_model":"qwen3-vl-32b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/qwen/qwen3-vl-8b-thinking":{"mode":"chat","base_model":"qwen3-vl-8b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/qwen/qwen3-vl-8b-instruct":{"mode":"chat","base_model":"qwen3-vl-8b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/google/gemini-2.5-flash-image":{"mode":"chat","base_model":"gemini-2.5-flash-image","deprecation_date":"2027-03-15","max_input_tokens":32768,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_tool_choice":false,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"openrouter/qwen/qwen3-vl-30b-a3b-thinking":{"mode":"chat","base_model":"qwen3-vl-30b-a3b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/qwen/qwen3-vl-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-vl-30b-a3b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/openai/gpt-5-pro":{"mode":"chat","base_model":"gpt-5-pro","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/qwen/qwen3-vl-235b-a22b-thinking":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-thinking","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/qwen/qwen3-vl-235b-a22b-instruct":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/qwen/qwen3-max":{"mode":"chat","base_model":"qwen3-max","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/deepseek/deepseek-v3.1-terminus":{"mode":"chat","base_model":"deepseek-v3.1-terminus","deprecation_date":"2026-09-28","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/qwen/qwen3-coder-flash":{"mode":"chat","base_model":"qwen3-coder-flash","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/qwen/qwen3-next-80b-a3b-thinking":{"mode":"chat","base_model":"qwen3-next-80b-a3b-thinking","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/qwen/qwen3-next-80b-a3b-instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","max_input_tokens":262144,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/qwen/qwen-plus-2025-07-28":{"mode":"chat","base_model":"qwen-plus-2025-07-28","max_input_tokens":1000000,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/moonshotai/kimi-k2-0905":{"mode":"chat","base_model":"kimi-k2-0905","max_input_tokens":262144,"max_output_tokens":98304,"max_tokens":98304,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":98304}}]},"openrouter/qwen/qwen3-30b-a3b-thinking-2507":{"mode":"chat","base_model":"qwen3-30b-a3b-thinking-2507","max_input_tokens":81920,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/mistralai/mistral-medium-3.1":{"mode":"chat","base_model":"mistral-medium-3.1","max_input_tokens":131072,"max_output_tokens":104857,"max_tokens":104857,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":104857}}]},"openrouter/z-ai/glm-4.5v":{"mode":"chat","base_model":"glm-4.5v","max_input_tokens":65536,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/qwen/qwen3-coder-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-coder-30b-a3b-instruct","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"together_ai/arcee-ai/trinity-mini":{"mode":"chat","base_model":"arcee-ai/trinity-mini","source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"vertex_ai/gemini-omni-1.1-flash":{"mode":"chat","base_model":"gemini-omni-1.1-flash","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"vertex_ai/gemini-omni-1.1-flash-preview":{"mode":"chat","base_model":"gemini-omni-1.1-flash","max_output_tokens":57920,"max_tokens":57920,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/v1beta/interactions"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text","video"],"supports_reasoning":true,"supports_video_input":true,"supports_vision":true,"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":57920}}]},"vertex_ai/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"gemma-4","source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","provider":"vertex_ai","max_output_tokens":8192,"max_tokens":8192,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"tpm":250000,"rpm":10,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"gpt-5.5-cyber":{"mode":"chat","base_model":"gpt-5.5-cyber","source":"https://developers.openai.com/api/docs/pricing","supports_reasoning":true,"provider":"openai","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"gpt-rosalind-research":{"mode":"chat","base_model":"gpt-rosalind-research","source":"https://developers.openai.com/api/docs/pricing","provider":"openai","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/meta-llama/Llama-3.1-405B-Instruct":{"mode":"chat","base_model":"llama-3.1-405b-instruct","source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/meta-llama/Llama-3.2-1B-Instruct":{"mode":"chat","base_model":"llama-3.2-1b-instruct","source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/meta-llama/Llama-3.2-3B-Instruct":{"mode":"chat","base_model":"llama-3.2-3b-instruct","source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/Qwen/Qwen2-1.5B-Instruct":{"mode":"chat","base_model":"qwen2-1.5b-instruct","source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/Qwen/Qwen2.5-14B-Instruct":{"mode":"chat","base_model":"qwen2.5-14b-instruct","source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"together_ai/Qwen/Qwen2.5-72B-Instruct":{"mode":"chat","base_model":"qwen2.5-72b-instruct","source":"https://api.together.ai/v1/models","provider":"together_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/codex-mini":{"mode":"chat","base_model":"codex-mini","deprecation_date":"2026-11-15","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/computer-use-preview":{"mode":"chat","base_model":"computer-use","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","deprecation_date":"2027-04-14","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-4.1-mini":{"mode":"chat","base_model":"gpt-4.1-mini","deprecation_date":"2027-04-14","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-4.1-nano":{"mode":"chat","base_model":"gpt-4.1-nano","deprecation_date":"2027-04-14","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-4o-2024-05-13":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-10-01","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5":{"mode":"chat","base_model":"gpt-5","deprecation_date":"2027-02-09","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5-codex":{"mode":"chat","base_model":"gpt-5-codex","deprecation_date":"2027-03-17","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","deprecation_date":"2027-02-09","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","deprecation_date":"2027-02-09","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5-pro":{"mode":"chat","base_model":"gpt-5-pro","deprecation_date":"2027-04-07","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5.1-codex-max":{"mode":"chat","base_model":"gpt-5.1-codex-max","deprecation_date":"2027-05-18","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5.2":{"mode":"chat","base_model":"gpt-5.2","deprecation_date":"2027-06-08","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5.2-codex":{"mode":"chat","base_model":"gpt-5.2-codex","deprecation_date":"2027-07-13","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5.2-pro":{"mode":"chat","base_model":"gpt-5.2-pro","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5.3-codex":{"mode":"chat","base_model":"gpt-5.3-codex","deprecation_date":"2027-08-24","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5.4-mini":{"mode":"chat","base_model":"gpt-5.4-mini","deprecation_date":"2027-09-21","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5.4-nano":{"mode":"chat","base_model":"gpt-5.4-nano","deprecation_date":"2027-09-21","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-5.4-pro":{"mode":"chat","base_model":"gpt-5.4-pro","deprecation_date":"2027-09-07","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/gpt-6-astra":{"mode":"chat","base_model":"gpt-6-astra","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","supports_prompt_cache_breakpoints":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/o1-mini":{"mode":"chat","base_model":"o1-mini","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/o3-2025-04-16":{"mode":"chat","base_model":"o3","deprecation_date":"2026-11-19","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/o3-deep-research":{"mode":"chat","base_model":"o3","deprecation_date":"2026-11-19","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/eu/o4-mini-2025-04-16":{"mode":"chat","base_model":"o4-mini","deprecation_date":"2026-11-19","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/codex-mini":{"mode":"chat","base_model":"codex-mini","deprecation_date":"2026-11-15","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/computer-use-preview":{"mode":"chat","base_model":"computer-use","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-4.1":{"mode":"chat","base_model":"gpt-4.1","deprecation_date":"2027-04-14","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-4.1-mini":{"mode":"chat","base_model":"gpt-4.1-mini","deprecation_date":"2027-04-14","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-4.1-nano":{"mode":"chat","base_model":"gpt-4.1-nano","deprecation_date":"2027-04-14","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-4o-2024-05-13":{"mode":"chat","base_model":"gpt-4o","deprecation_date":"2026-10-01","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5":{"mode":"chat","base_model":"gpt-5","deprecation_date":"2027-02-09","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5-codex":{"mode":"chat","base_model":"gpt-5-codex","deprecation_date":"2027-03-17","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5-mini":{"mode":"chat","base_model":"gpt-5-mini","deprecation_date":"2027-02-09","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","deprecation_date":"2027-02-09","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5-pro":{"mode":"chat","base_model":"gpt-5-pro","deprecation_date":"2027-04-07","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5.1-codex-max":{"mode":"chat","base_model":"gpt-5.1-codex-max","deprecation_date":"2027-05-18","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5.2":{"mode":"chat","base_model":"gpt-5.2","deprecation_date":"2027-06-08","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5.2-codex":{"mode":"chat","base_model":"gpt-5.2-codex","deprecation_date":"2027-07-13","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5.2-pro":{"mode":"chat","base_model":"gpt-5.2-pro","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5.3-codex":{"mode":"chat","base_model":"gpt-5.3-codex","deprecation_date":"2027-08-24","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5.4-mini":{"mode":"chat","base_model":"gpt-5.4-mini","deprecation_date":"2027-09-21","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5.4-nano":{"mode":"chat","base_model":"gpt-5.4-nano","deprecation_date":"2027-09-21","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/gpt-5.4-pro":{"mode":"chat","base_model":"gpt-5.4-pro","deprecation_date":"2027-09-07","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/o1-mini":{"mode":"chat","base_model":"o1-mini","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"azure/us/o3-deep-research":{"mode":"chat","base_model":"o3","deprecation_date":"2026-11-19","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"aihubmix/agnes-2.5-flash":{"mode":"chat","base_model":"agnes-2.5-flash","max_input_tokens":512000,"max_output_tokens":65500,"max_tokens":65500,"source":"https://aihubmix.com/api/v1/models","supports_reasoning":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65500}}]},"aihubmix/agnes-2.5-pro":{"mode":"chat","base_model":"agnes-2.5-pro","max_input_tokens":1000000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/cc-glm-5.1":{"mode":"chat","base_model":"cc-glm-5.1","max_input_tokens":200000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"prompt_cache_min_tokens":512,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_sampling_params":false,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"prompt_cache_min_tokens":4096,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"aihubmix/claude-opus-4-8-think":{"mode":"chat","base_model":"claude-opus-4-8-think","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"supports_adaptive_thinking":true,"supports_sampling_params":false,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"supports_adaptive_thinking":true,"prompt_cache_min_tokens":512,"supports_sampling_params":false,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"prompt_cache_min_tokens":1024,"supports_adaptive_thinking":true,"supports_sampling_params":false,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/coding-glm-5.3":{"mode":"chat","base_model":"coding-glm-5.3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/coding-kimi-k3":{"mode":"chat","base_model":"coding-kimi-k3","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"aihubmix/coding-xiaomi-mimo-v2-omni":{"mode":"chat","base_model":"coding-xiaomi-mimo-v2-omni","source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"aihubmix/coding-xiaomi-mimo-v2.5":{"mode":"chat","base_model":"coding-xiaomi-mimo","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/coding-xiaomi-mimo-v2.5-pro":{"mode":"chat","base_model":"coding-xiaomi-mimo-v2.5-pro","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/command-a-plus-05-2026":{"mode":"chat","base_model":"command-a-plus","max_input_tokens":128000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"aihubmix/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"aihubmix/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":384000,"max_tokens":384000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"aihubmix/doubao-seed-2-0-code-preview":{"mode":"chat","base_model":"doubao-seed-2-0-code","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/doubao-seed-2-0-lite-260428":{"mode":"chat","base_model":"doubao-seed-2-0-lite","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/doubao-seed-2-0-mini":{"mode":"chat","base_model":"doubao-seed-2-0-mini","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/doubao-seed-2-0-pro":{"mode":"chat","base_model":"doubao-seed-2-0-pro","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/doubao-seed-2-1-turbo":{"mode":"chat","base_model":"doubao-seed-2-1-turbo","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"aihubmix/ernie-5.1":{"mode":"chat","base_model":"ernie-5.1","max_input_tokens":119000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_prompt_caching":true,"supports_reasoning":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/gemini-3-flash-preview":{"mode":"chat","base_model":"gemini-3-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/gemini-3-flash-preview-search":{"mode":"chat","base_model":"gemini-3-flash-search","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/gemini-3.1-pro-preview":{"mode":"chat","base_model":"gemini-3.1-pro","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/gemini-3.1-pro-preview-customtools":{"mode":"chat","base_model":"gemini-3.1-pro-customtools","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/gemini-3.5-flash-lite":{"mode":"chat","base_model":"gemini-3.5-flash-lite","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/gemini-3.7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"gemma-4-26b-a4b-it","max_input_tokens":262144,"max_output_tokens":131100,"max_tokens":131100,"source":"https://aihubmix.com/api/v1/models","supports_reasoning":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131100}}]},"aihubmix/gemma-4-31b-it":{"mode":"chat","base_model":"gemma-4-31b-it","max_input_tokens":262144,"max_output_tokens":131100,"max_tokens":131100,"source":"https://aihubmix.com/api/v1/models","supports_reasoning":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131100}}]},"aihubmix/glm-5.2-fast-preview":{"mode":"chat","base_model":"glm-5.2-fast","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/glm-5.3":{"mode":"chat","base_model":"glm-5.3","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/glm-5.3-flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/glm-5v-turbo":{"mode":"chat","base_model":"glm-5v-turbo","max_input_tokens":200000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/gpt-5.3-codex":{"mode":"chat","base_model":"gpt-5.3-codex","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.4-high":{"mode":"chat","base_model":"gpt-5.4-high","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.4-low":{"mode":"chat","base_model":"gpt-5.4-low","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.4-mini":{"mode":"chat","base_model":"gpt-5.4-mini","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.4-nano":{"mode":"chat","base_model":"gpt-5.4-nano","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.5":{"mode":"chat","base_model":"gpt-5.5","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.5-pro":{"mode":"chat","base_model":"gpt-5.5-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.6-sol-disc":{"mode":"chat","base_model":"gpt-5.6-sol-disc","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/gpt-chat-latest":{"mode":"chat","base_model":"gpt-chat","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/grok-4-20-non-reasoning":{"mode":"chat","base_model":"grok-4-20-non","max_input_tokens":1000000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"aihubmix/grok-4-20-reasoning":{"mode":"chat","base_model":"grok-4-20","max_input_tokens":1000000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"aihubmix/grok-4.6":{"mode":"chat","base_model":"grok-4.6","max_input_tokens":500000,"max_output_tokens":500000,"max_tokens":500000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":500000}}]},"aihubmix/grok-build-0.1":{"mode":"chat","base_model":"grok-build-0.1","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"aihubmix/hy3":{"mode":"chat","base_model":"hy3","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"aihubmix/hy4-preview":{"mode":"chat","base_model":"hy4","max_input_tokens":1048576,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"aihubmix/kimi-k2.6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"aihubmix/kimi-k2.7-code-highspeed":{"mode":"chat","base_model":"kimi-k2.7-code-highspeed","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"aihubmix/kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"aihubmix/longcat-2.0":{"mode":"chat","base_model":"longcat-2.0","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/mai-thinking-1":{"mode":"chat","base_model":"mai-thinking-1","max_input_tokens":256000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aihubmix.com/api/v1/models","supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"aihubmix/mimo-v2-omni":{"mode":"chat","base_model":"mimo-v2-omni","max_input_tokens":256000,"source":"https://aihubmix.com/api/v1/models","supports_prompt_caching":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"aihubmix/mimo-v2-pro":{"mode":"chat","base_model":"mimo-v2-pro","max_input_tokens":1000000,"source":"https://aihubmix.com/api/v1/models","supports_prompt_caching":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"aihubmix/minimax-m2.7":{"mode":"chat","base_model":"minimax-m2.7","max_input_tokens":204800,"max_output_tokens":204800,"max_tokens":204800,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"aihubmix/minimax-m3":{"mode":"chat","base_model":"minimax-m3","max_input_tokens":1000000,"max_output_tokens":524288,"max_tokens":524288,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"aihubmix/muse-spark-1.2":{"mode":"chat","base_model":"muse-spark-1.2","max_input_tokens":1048576,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"aihubmix/qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_response_schema":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/qwen3.5-122b-a10b":{"mode":"chat","base_model":"qwen3.5-122b-a10b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/qwen3.5-397b-a17b":{"mode":"chat","base_model":"qwen3.5-397b-a17b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/qwen3.6-27b":{"mode":"chat","base_model":"qwen3.6-27b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/qwen3.6-35b-a3b":{"mode":"chat","base_model":"qwen3.6-35b-a3b","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/qwen3.6-max-preview":{"mode":"chat","base_model":"qwen3.6-max","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"aihubmix/qwen3.7-plus":{"mode":"chat","base_model":"qwen3.7-plus","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/qwen3.8-2.4t-a95b":{"mode":"chat","base_model":"qwen3.8-2.4t-a95b","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/qwen3.8-flash":{"mode":"chat","base_model":"qwen3.8-flash","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/qwen3.8-max":{"mode":"chat","base_model":"qwen3.8-max","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"source":"https://aihubmix.com/api/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_web_search":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"aihubmix/step-3.7-flash":{"mode":"chat","base_model":"step-3.7-flash","max_input_tokens":256000,"source":"https://aihubmix.com/api/v1/models","supports_prompt_caching":true,"supports_reasoning":true,"supports_vision":true,"provider":"aihubmix","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/typesafe/jev-1.13":{"mode":"chat","base_model":"jev-1.13","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800,"source":"https://openrouter.ai/typesafe/jev-1.13","provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":28800}}]},"wandb/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1049000,"source":"https://wandb.ai/site/pricing/tokens/","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"wandb","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/~anthropic/claude-fable-latest":{"mode":"chat","base_model":"claude-fable-latest","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/~anthropic/claude-haiku-latest":{"mode":"chat","base_model":"claude-haiku-latest","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"openrouter/~anthropic/claude-opus-latest":{"mode":"chat","base_model":"claude-opus-latest","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/~anthropic/claude-sonnet-latest":{"mode":"chat","base_model":"claude-sonnet-latest","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/~deepseek/deepseek-flash-latest":{"mode":"chat","base_model":"deepseek-flash-latest","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/~deepseek/deepseek-pro-latest":{"mode":"chat","base_model":"deepseek-pro-latest","max_input_tokens":1048576,"max_output_tokens":384000,"max_tokens":384000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"openrouter/~deepseek/deepseek-v4-flash-latest":{"mode":"chat","base_model":"deepseek-v4-flash-latest","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/~google/gemini-flash-latest":{"mode":"chat","base_model":"gemini-flash-latest","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/~google/gemini-pro-latest":{"mode":"chat","base_model":"gemini-pro-latest","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/~moonshotai/kimi-latest":{"mode":"chat","base_model":"kimi-latest","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/~openai/gpt-astra-latest":{"mode":"chat","base_model":"gpt-astra-latest","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/~openai/gpt-luna-latest":{"mode":"chat","base_model":"gpt-luna-latest","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/~openai/gpt-mini-latest":{"mode":"chat","base_model":"gpt-mini-latest","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/~openai/gpt-sol-latest":{"mode":"chat","base_model":"gpt-sol-latest","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/~openai/gpt-terra-latest":{"mode":"chat","base_model":"gpt-terra-latest","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/~x-ai/grok-latest":{"mode":"chat","base_model":"grok-latest","max_input_tokens":500000,"max_output_tokens":450000,"max_tokens":450000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":450000}}]},"openrouter/~z-ai/glm-flash-latest":{"mode":"chat","base_model":"glm-flash-latest","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/~z-ai/glm-latest":{"mode":"chat","base_model":"glm-latest","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/aion-labs/aion-2.0":{"mode":"chat","base_model":"aion-2.0","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/aion-labs/aion-3.0":{"mode":"chat","base_model":"aion-3.0","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/aion-labs/aion-3.0-mini":{"mode":"chat","base_model":"aion-3.0-mini","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/amazon/nova-2-lite-v1":{"mode":"chat","base_model":"nova-2-lite-v1","max_input_tokens":1000000,"max_output_tokens":65535,"max_tokens":65535,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65535}}]},"openrouter/amazon/nova-premier-v1":{"mode":"chat","base_model":"nova-premier-v1","max_input_tokens":1000000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"openrouter/anthropic/claude-fable-5:batch":{"mode":"chat","base_model":"claude-fable-5:batch","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-fable-5.1:batch":{"mode":"chat","base_model":"claude-fable-5.1:batch","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-haiku-4.5:batch":{"mode":"chat","base_model":"claude-haiku-4.5:batch","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"openrouter/anthropic/claude-opus-4.1:batch":{"mode":"chat","base_model":"claude-opus-4.1:batch","max_input_tokens":200000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"openrouter/anthropic/claude-opus-4.5:batch":{"mode":"chat","base_model":"claude-opus-4.5:batch","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"openrouter/anthropic/claude-opus-4.6:batch":{"mode":"chat","base_model":"claude-opus-4.6:batch","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-4.7:batch":{"mode":"chat","base_model":"claude-opus-4.7:batch","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-4.8:batch":{"mode":"chat","base_model":"claude-opus-4.8:batch","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-5:batch":{"mode":"chat","base_model":"claude-opus-5:batch","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-sonnet-4.5:batch":{"mode":"chat","base_model":"claude-sonnet-4.5:batch","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"openrouter/anthropic/claude-sonnet-4.6:batch":{"mode":"chat","base_model":"claude-sonnet-4.6:batch","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-sonnet-5:batch":{"mode":"chat","base_model":"claude-sonnet-5:batch","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/arcee-ai/trinity-large-thinking":{"mode":"chat","base_model":"trinity-large-thinking","max_input_tokens":262144,"max_output_tokens":80000,"max_tokens":80000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":80000}}]},"openrouter/baidu/ernie-4.5-vl-424b-a47b":{"mode":"chat","base_model":"ernie-4.5-vl-424b-a47b","deprecation_date":"2026-10-08","max_input_tokens":123000,"max_output_tokens":16000,"max_tokens":16000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16000}}]},"openrouter/bytedance-seed/seed-1.6":{"mode":"chat","base_model":"seed-1.6","deprecation_date":"2026-11-11","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/bytedance-seed/seed-1.6-flash":{"mode":"chat","base_model":"seed-1.6-flash","deprecation_date":"2026-11-11","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/bytedance-seed/seed-2-1-turbo":{"mode":"chat","base_model":"seed-2-1-turbo","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/bytedance-seed/seed-2.0-code":{"mode":"chat","base_model":"seed-2.0-code","deprecation_date":"2026-11-11","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/bytedance-seed/seed-2.0-lite":{"mode":"chat","base_model":"seed-2.0-lite","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/bytedance-seed/seed-2.0-mini":{"mode":"chat","base_model":"seed-2.0-mini","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/cognitivecomputations/dolphin-mistral-24b-venice-edition":{"mode":"chat","base_model":"dolphin-mistral-24b-venice-edition","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"openrouter/cohere/north-mini-code:free":{"mode":"chat","base_model":"north-mini-code","max_input_tokens":256000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"openrouter/dots-studio/dots-3-note-preview:free":{"mode":"chat","base_model":"dots-studio/dots-3-note","deprecation_date":"2026-12-31","max_input_tokens":512000,"max_output_tokens":460800,"max_tokens":460800,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":460800}}]},"openrouter/google/gemini-2.5-flash-lite:batch":{"mode":"chat","base_model":"gemini-2.5-flash-lite:batch","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65535}}]},"openrouter/google/gemini-2.5-flash:batch":{"mode":"chat","base_model":"gemini-2.5-flash:batch","deprecation_date":"2026-10-20","max_input_tokens":1048576,"max_output_tokens":65535,"max_tokens":65535,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65535}}]},"openrouter/google/gemini-2.5-pro:batch":{"mode":"chat","base_model":"gemini-2.5-pro:batch","deprecation_date":"2026-10-20","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3-flash-preview:batch":{"mode":"chat","base_model":"gemini-3-flash-preview:batch","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.1-flash-lite:batch":{"mode":"chat","base_model":"gemini-3.1-flash-lite:batch","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.1-pro-preview:batch":{"mode":"chat","base_model":"gemini-3.1-pro-preview:batch","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.5-flash-lite:batch":{"mode":"chat","base_model":"gemini-3.5-flash-lite:batch","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.5-flash:batch":{"mode":"chat","base_model":"gemini-3.5-flash:batch","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.6-flash:batch":{"mode":"chat","base_model":"gemini-3.6-flash:batch","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.7-flash:batch":{"mode":"chat","base_model":"gemini-3.7-flash:batch","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/google/gemini-3.8-flash:batch":{"mode":"chat","base_model":"gemini-3.8-flash:batch","max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_video_input":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/ibm-granite/granite-4.0-h-micro":{"mode":"chat","base_model":"granite-4.0-h-micro","max_input_tokens":131000,"max_output_tokens":117900,"max_tokens":117900,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":117900}}]},"openrouter/ibm-granite/granite-4.2-8b":{"mode":"chat","base_model":"granite-4.2-8b","max_input_tokens":131072,"max_output_tokens":117964,"max_tokens":117964,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":117964}}]},"openrouter/inception/mercury-2":{"mode":"chat","base_model":"mercury-2","max_input_tokens":128000,"max_output_tokens":50000,"max_tokens":50000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":50000}}]},"openrouter/inception/mercury-2.5":{"mode":"chat","base_model":"mercury-2.5","max_input_tokens":260000,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/inclusionai/ling-3.0-flash":{"mode":"chat","base_model":"ling-3.0-flash","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/inclusionai/ling-3.0-flash-fin":{"mode":"chat","base_model":"ling-3.0-flash-fin","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/inclusionai/ling-3.0-flash-fin:free":{"mode":"chat","base_model":"inclusionai/ling-3.0-flash-fin","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/inclusionai/ling-3.0-flash-sante:free":{"mode":"chat","base_model":"inclusionai/ling-3.0-flash-sante","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/inclusionai/ling-3.0-flash-vl":{"mode":"chat","base_model":"ling-3.0-flash-vl","max_input_tokens":131072,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/inclusionai/ling-3.0-flash-vl:free":{"mode":"chat","base_model":"inclusionai/ling-3.0-flash-vl","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/inference-net/schematron-v2-small":{"mode":"chat","base_model":"schematron-v2-small","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"openrouter/inference-net/schematron-v2-turbo":{"mode":"chat","base_model":"schematron-v2-turbo","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"openrouter/kwaipilot/kat-coder-pro-v2.5":{"mode":"chat","base_model":"kat-coder-pro-v2.5","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/liquid/lfm-2.5-2.6b:free":{"mode":"chat","base_model":"liquid/lfm-2.5-2.6b","max_input_tokens":65536,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"openrouter/meituan/longcat-2.0":{"mode":"chat","base_model":"longcat-2.0","max_input_tokens":1048756,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"openrouter/meta/muse-glimmer-30b":{"mode":"chat","base_model":"muse-glimmer-30b","max_input_tokens":131072,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/meta/muse-spark-1.1":{"mode":"chat","base_model":"muse-spark-1.1","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/meta/muse-spark-1.2":{"mode":"chat","base_model":"muse-spark-1.2","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/meta/muse-spark-1.2-contributor":{"mode":"chat","base_model":"muse-spark-1.2-contributor","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/meta/muse-spark-1.3":{"mode":"chat","base_model":"muse-spark-1.3","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/meta/muse-spark-1.3-contributor":{"mode":"chat","base_model":"muse-spark-1.3-contributor","max_input_tokens":1048576,"max_output_tokens":943718,"max_tokens":943718,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":943718}}]},"openrouter/mistralai/codestral-2508:batch":{"mode":"chat","base_model":"codestral-2508:batch","max_input_tokens":256000,"max_output_tokens":204800,"max_tokens":204800,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"openrouter/mistralai/ministral-8b-2512:batch":{"mode":"chat","base_model":"ministral-8b-2512:batch","max_input_tokens":262144,"max_output_tokens":209715,"max_tokens":209715,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":209715}}]},"openrouter/mistralai/mistral-large-2512:batch":{"mode":"chat","base_model":"mistral-large-2512:batch","max_input_tokens":262144,"max_output_tokens":209715,"max_tokens":209715,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":209715}}]},"openrouter/mistralai/mistral-medium-3-5:batch":{"mode":"chat","base_model":"mistral-medium-3-5:batch","max_input_tokens":262144,"max_output_tokens":209715,"max_tokens":209715,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":209715}}]},"openrouter/mistralai/mistral-medium-3.1:batch":{"mode":"chat","base_model":"mistral-medium-3.1:batch","max_input_tokens":131072,"max_output_tokens":104857,"max_tokens":104857,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":104857}}]},"openrouter/mistralai/mistral-small-2603:batch":{"mode":"chat","base_model":"mistral-small-2603:batch","max_input_tokens":262144,"max_output_tokens":209715,"max_tokens":209715,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":209715}}]},"openrouter/moonshotai/kimi-k3:batch":{"mode":"chat","base_model":"kimi-k3:batch","max_input_tokens":1048576,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/nex-agi/nex-n2.5-mini:free":{"mode":"chat","base_model":"nex-agi/nex-n2.5-mini","deprecation_date":"2026-09-25","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/nex-agi/nex-n2.5-pro:free":{"mode":"chat","base_model":"nex-agi/nex-n2.5-pro","deprecation_date":"2026-09-25","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/nousresearch/hermes-4-405b":{"mode":"chat","base_model":"hermes-4-405b","max_input_tokens":131072,"max_output_tokens":117964,"max_tokens":117964,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":117964}}]},"openrouter/openai/gpt-3.5-turbo:batch":{"mode":"chat","base_model":"gpt-3.5-turbo:batch","max_input_tokens":16385,"max_output_tokens":4096,"max_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"openrouter/openai/gpt-4-turbo:batch":{"mode":"chat","base_model":"gpt-4-turbo:batch","max_input_tokens":128000,"max_output_tokens":4096,"max_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"openrouter/openai/gpt-4.1-mini:batch":{"mode":"chat","base_model":"gpt-4.1-mini:batch","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/openai/gpt-4.1-nano:batch":{"mode":"chat","base_model":"gpt-4.1-nano:batch","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/openai/gpt-4.1:batch":{"mode":"chat","base_model":"gpt-4.1:batch","max_input_tokens":1047576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/openai/gpt-4o-mini:batch":{"mode":"chat","base_model":"gpt-4o-mini:batch","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/openai/gpt-4o:batch":{"mode":"chat","base_model":"gpt-4o:batch","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/openai/gpt-5-image":{"mode":"chat","base_model":"gpt-5-image","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5-image-mini":{"mode":"chat","base_model":"gpt-5-image-mini","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5-mini:batch":{"mode":"chat","base_model":"gpt-5-mini:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5-nano:batch":{"mode":"chat","base_model":"gpt-5-nano:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5-pro:batch":{"mode":"chat","base_model":"gpt-5-pro:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5:batch":{"mode":"chat","base_model":"gpt-5:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.1:batch":{"mode":"chat","base_model":"gpt-5.1:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.2-pro:batch":{"mode":"chat","base_model":"gpt-5.2-pro:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.2:batch":{"mode":"chat","base_model":"gpt-5.2:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.4-image-2":{"mode":"chat","base_model":"gpt-5.4-image-2","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.4-mini:batch":{"mode":"chat","base_model":"gpt-5.4-mini:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.4-nano:batch":{"mode":"chat","base_model":"gpt-5.4-nano:batch","max_input_tokens":400000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.4-pro:batch":{"mode":"chat","base_model":"gpt-5.4-pro:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.4:batch":{"mode":"chat","base_model":"gpt-5.4:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.5-pro:batch":{"mode":"chat","base_model":"gpt-5.5-pro:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.5:batch":{"mode":"chat","base_model":"gpt-5.5:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-luna-pro:batch":{"mode":"chat","base_model":"gpt-5.6-luna-pro:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-luna:batch":{"mode":"chat","base_model":"gpt-5.6-luna:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-sol-pro:batch":{"mode":"chat","base_model":"gpt-5.6-sol-pro:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-sol:batch":{"mode":"chat","base_model":"gpt-5.6-sol:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-terra-pro:batch":{"mode":"chat","base_model":"gpt-5.6-terra-pro:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-5.6-terra:batch":{"mode":"chat","base_model":"gpt-5.6-terra:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-astra-pro:batch":{"mode":"chat","base_model":"gpt-6-astra-pro:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-astra:batch":{"mode":"chat","base_model":"gpt-6-astra:batch","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-luna":{"mode":"chat","base_model":"gpt-6-luna","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-luna-pro":{"mode":"chat","base_model":"gpt-6-luna-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-sol":{"mode":"chat","base_model":"gpt-6-sol","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-sol-pro":{"mode":"chat","base_model":"gpt-6-sol-pro","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/o3-mini:batch":{"mode":"chat","base_model":"o3-mini:batch","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":100000}}]},"openrouter/openai/o3:batch":{"mode":"chat","base_model":"o3:batch","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":100000}}]},"openrouter/openai/o4-mini:batch":{"mode":"chat","base_model":"o4-mini:batch","max_input_tokens":200000,"max_output_tokens":100000,"max_tokens":100000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":100000}}]},"openrouter/perceptron/perceptron-mk1":{"mode":"chat","base_model":"perceptron-mk1","max_input_tokens":32768,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"openrouter/perplexity/sonar-pro-search":{"mode":"chat","base_model":"sonar-pro-search","max_input_tokens":200000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"openrouter/qwen/qwen3.8-27b:free":{"mode":"chat","base_model":"qwen3.8-27b","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/rekaai/reka-edge":{"mode":"chat","base_model":"reka-edge","max_input_tokens":16384,"max_output_tokens":14745,"max_tokens":14745,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":14745}}]},"openrouter/rekaai/reka-flash-3":{"mode":"chat","base_model":"reka-flash-3","max_input_tokens":65536,"max_output_tokens":58982,"max_tokens":58982,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":58982}}]},"openrouter/relace/relace-apply-3":{"mode":"chat","base_model":"relace-apply-3","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/relace/relace-search":{"mode":"chat","base_model":"relace-search","max_input_tokens":256000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/sakana/fugu-max":{"mode":"chat","base_model":"fugu-max","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/sakana/fugu-ultra":{"mode":"chat","base_model":"fugu-ultra","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/sakana/fugu-ultra-v2":{"mode":"chat","base_model":"fugu-ultra-v2","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/sakana/sakana-namazu":{"mode":"chat","base_model":"sakana-namazu","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/stepfun/step-3.5-flash":{"mode":"chat","base_model":"step-3.5-flash","max_input_tokens":262144,"max_output_tokens":65536,"max_tokens":65536,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"openrouter/stepfun/step-3.7-flash":{"mode":"chat","base_model":"step-3.7-flash","max_input_tokens":256000,"max_output_tokens":230400,"max_tokens":230400,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":230400}}]},"openrouter/tencent/hy-mt2-1.8b":{"mode":"chat","base_model":"hy-mt2-1.8b","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"openrouter/tencent/hy-mt2-30b-a3b":{"mode":"chat","base_model":"hy-mt2-30b-a3b","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"openrouter/tencent/hy-mt2-7b":{"mode":"chat","base_model":"hy-mt2-7b","max_input_tokens":8192,"max_output_tokens":4096,"max_tokens":4096,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"openrouter/tencent/hy3":{"mode":"chat","base_model":"hy3","max_input_tokens":262144,"max_output_tokens":128000,"max_tokens":128000,"off_peak_pricing":{"hours_utc":"16:00-00:00","input_cost_per_token":8.25e-8,"output_cost_per_token":3.3e-7,"cache_read_input_token_cost":2.0625e-8},"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/tencent/hy3-preview":{"mode":"chat","base_model":"hy3-preview","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/tencent/hy4-preview":{"mode":"chat","base_model":"hy4-preview","max_input_tokens":1048576,"max_output_tokens":64000,"max_tokens":64000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"openrouter/thedrummer/cydonia-24b-v4.1":{"mode":"chat","base_model":"cydonia-24b-v4.1","max_input_tokens":131072,"max_output_tokens":117964,"max_tokens":117964,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":117964}}]},"openrouter/thinkingmachines/inkling":{"mode":"chat","base_model":"inkling","max_input_tokens":524288,"max_output_tokens":471859,"max_tokens":471859,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":471859}}]},"openrouter/thinkingmachines/inkling-small":{"mode":"chat","base_model":"inkling-small","max_input_tokens":524288,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"openrouter/thinkingmachines/inkling-small:free":{"mode":"chat","base_model":"thinkingmachines/inkling-small","max_input_tokens":1048576,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"openrouter/thinkingmachines/inkling:free":{"mode":"chat","base_model":"thinkingmachines/inkling","max_input_tokens":1048576,"max_output_tokens":262144,"max_tokens":262144,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"openrouter/unbiased/pareto":{"mode":"chat","base_model":"pareto","max_input_tokens":262144,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/upstage/solar-pro-3":{"mode":"chat","base_model":"solar-pro-3","max_input_tokens":131072,"max_output_tokens":117964,"max_tokens":117964,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":117964}}]},"openrouter/upstage/solar-pro4":{"mode":"chat","base_model":"solar-pro4","max_input_tokens":524288,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/writer/palmyra-x5":{"mode":"chat","base_model":"palmyra-x5","max_input_tokens":1040000,"max_output_tokens":8192,"max_tokens":8192,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":false,"supports_tool_choice":false,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"openrouter/x-ai/grok-4.3:batch":{"mode":"chat","base_model":"grok-4.3:batch","max_input_tokens":1000000,"max_output_tokens":900000,"max_tokens":900000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":900000}}]},"openrouter/z-ai/glm-5.3-flash:batch":{"mode":"chat","base_model":"glm-5.3-flash:batch","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/z-ai/glm-5.3:batch":{"mode":"chat","base_model":"glm-5.3:batch","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/prism-ml/ternary-bonsai-2-27b":{"mode":"chat","base_model":"ternary-bonsai-2-27b","max_input_tokens":262144,"max_output_tokens":32768,"max_tokens":32768,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"openrouter/z-ai/glm-5.3-flashx":{"mode":"chat","base_model":"glm-5.3-flashx","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/x-ai/grok-4.7":{"mode":"chat","base_model":"grok-4.7","max_input_tokens":500000,"max_output_tokens":450000,"max_tokens":450000,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":450000}}]},"openrouter/xiaomi/mimo-v2.6-flash":{"mode":"chat","base_model":"mimo-v2.6-flash","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/xiaomi/mimo-v2.6-pro":{"mode":"chat","base_model":"mimo-v2.6-pro","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/xiaomi/mimo-v2.6-pro-ultraspeed":{"mode":"chat","base_model":"mimo-v2.6-pro-ultraspeed","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":true,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"xiaomi_mimo/mimo-v2.6-pro":{"mode":"chat","base_model":"mimo-v2.6-pro","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.xiaomimimo.com/static/docs/price/pay-as-you-go.md","supported_endpoints":["/v1/chat/completions"],"supports_audio_input":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"xiaomi_mimo","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"xiaomi_mimo/mimo-v2.6-flash":{"mode":"chat","base_model":"mimo-v2.6-flash","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://platform.xiaomimimo.com/static/docs/price/pay-as-you-go.md","supported_endpoints":["/v1/chat/completions"],"supports_audio_input":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"xiaomi_mimo","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"xai/grok-4.20-0309":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta-0309":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta-latest":{"mode":"chat","base_model":"grok-4.20-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta-latest-non-reasoning":{"mode":"chat","base_model":"grok-4.20-beta-latest-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta-latest-reasoning":{"mode":"chat","base_model":"grok-4.20-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta-non-reasoning":{"mode":"chat","base_model":"grok-4.20-beta-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-beta-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-experimental-beta-0304":{"mode":"chat","base_model":"grok-4.20-experimental","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-experimental-beta-0304-non-reasoning":{"mode":"chat","base_model":"grok-4.20-experimental-beta-0304-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-experimental-beta-0304-reasoning":{"mode":"chat","base_model":"grok-4.20-experimental-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-experimental-beta-latest":{"mode":"chat","base_model":"grok-4.20-experimental-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-experimental-beta-non-reasoning-latest":{"mode":"chat","base_model":"grok-4.20-experimental-beta-non","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-experimental-beta-reasoning-latest":{"mode":"chat","base_model":"grok-4.20-experimental-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-multi-agent-beta-latest":{"mode":"responses","base_model":"grok-4.20-multi-agent-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"supported_endpoints":["/v1/responses"],"provider":"xai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-multi-agent-experimental-beta-0304":{"mode":"responses","base_model":"grok-4.20-multi-agent-experimental","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"supported_endpoints":["/v1/responses"],"provider":"xai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-multi-agent-experimental-beta-latest":{"mode":"responses","base_model":"grok-4.20-multi-agent-experimental-beta","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"supported_endpoints":["/v1/responses"],"provider":"xai","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-non-reasoning-gv2":{"mode":"chat","base_model":"grok-4.20-non-reasoning-gv2","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_prompt_caching":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"xai/grok-4.20-reasoning-gv2":{"mode":"chat","base_model":"grok-4.20-reasoning-gv2","max_input_tokens":1000000,"max_output_tokens":1000000,"max_tokens":1000000,"source":"https://api.x.ai/v1/language-models","supports_function_calling":true,"supports_reasoning":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"supports_prompt_caching":true,"supports_response_schema":true,"provider":"xai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"openrouter/nex-agi/nex-n2.5-mini":{"mode":"chat","base_model":"nex-n2.5-mini","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":false,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"openrouter/nex-agi/nex-n2.5-pro":{"mode":"chat","base_model":"nex-n2.5-pro","max_input_tokens":262144,"max_output_tokens":235929,"max_tokens":235929,"source":"https://openrouter.ai/api/v1/models","supports_audio_input":false,"supports_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":false,"provider":"openrouter","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":235929}}]},"baseten/deepseek-ai/DeepSeek-V4.1-Flash":{"mode":"chat","base_model":"deepseek-v4.1-flash","max_input_tokens":1048576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"baseten/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262000,"max_output_tokens":262000,"max_tokens":262000,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262000}}]},"baseten/moonshotai/Kimi-K2.7-Code":{"mode":"chat","base_model":"kimi-k2.7-code","max_input_tokens":262000,"max_output_tokens":262000,"max_tokens":262000,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262000}}]},"baseten/moonshotai/Kimi-K3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_output_tokens":262144,"max_tokens":262144,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"baseten/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"mode":"chat","base_model":"nvidia-nemotron-3-ultra-550b-a55b","max_input_tokens":202800,"max_output_tokens":202800,"max_tokens":202800,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202800}}]},"baseten/thinkingmachines/inkling":{"mode":"chat","base_model":"thinkingmachines/inkling","max_input_tokens":1048576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"baseten/thinkingmachines/inkling-small":{"mode":"chat","base_model":"thinkingmachines/inkling-small","max_input_tokens":1048576,"max_output_tokens":32768,"max_tokens":32768,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"baseten/zai-org/GLM-5.2":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1048576,"max_output_tokens":262144,"max_tokens":262144,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"baseten/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"glm-5.3-flash","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"baseten/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1048576,"max_output_tokens":384000,"max_tokens":384000,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":384000}}]},"baseten/deepseek-ai/DeepSeek-V4-Pro":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1048576,"max_output_tokens":262144,"max_tokens":262144,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"baseten/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1048576,"max_output_tokens":262144,"max_tokens":262144,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":false,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"baseten/zai-org/GLM-5.2-Fast":{"mode":"chat","base_model":"glm-5.2-fast","max_input_tokens":1048576,"max_output_tokens":262144,"max_tokens":262144,"source":"https://inference.baseten.co/v1/models","supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"baseten","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"vertex_ai/gemini-2.0-flash":{"mode":"chat","base_model":"gemini-2.0-flash","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"vertex_ai/gemini-2.0-flash-lite":{"mode":"chat","base_model":"gemini-2.0-flash-lite","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"vertex_ai/zai-org/glm-5.2-maas":{"mode":"chat","base_model":"glm-5.2","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_regions":["global"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"openrouter/cohere/command-a-plus":{"mode":"chat","base_model":"command-a-plus","provider":"openrouter","max_input_tokens":192000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"openrouter/openai/gpt-6-luna-pro:batch":{"mode":"chat","base_model":"gpt-6-luna-pro:batch","provider":"openrouter","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-luna:batch":{"mode":"chat","base_model":"gpt-6-luna:batch","provider":"openrouter","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-sol-pro:batch":{"mode":"chat","base_model":"gpt-6-sol-pro:batch","provider":"openrouter","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/openai/gpt-6-sol:batch":{"mode":"chat","base_model":"gpt-6-sol:batch","provider":"openrouter","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/anthropic/claude-opus-5.5:batch":{"mode":"chat","base_model":"claude-opus-5.5:batch","provider":"openrouter","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_pdf_input":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"openrouter/assemblyai/universal-3-5-pro":{"mode":"chat","base_model":"universal-3-5-pro","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/qwen/qwen3.8-omni-flash":{"mode":"chat","base_model":"qwen3.8-omni-flash","provider":"openrouter","max_input_tokens":1000000,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_audio_input":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/~typesafe/jev-latest":{"mode":"chat","base_model":"jev-latest","provider":"openrouter","max_input_tokens":32000,"max_output_tokens":28800,"max_tokens":28800,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":28800}}]},"openrouter/meta/muse-voice-transcribe-1.0":{"mode":"chat","base_model":"muse-voice-transcribe-1.0","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/deepseek/deepseek-v4.1-flash:batch":{"mode":"chat","base_model":"deepseek-v4.1-flash:batch","provider":"openrouter","max_input_tokens":1048576,"max_output_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"openrouter/microsoft/mai-transcribe-2":{"mode":"chat","base_model":"mai-transcribe-2","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/nvidia/nemotron-3.5-asr-streaming-multilingual-0.6b":{"mode":"chat","base_model":"nemotron-3.5-asr-streaming-multilingual-0.6b","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/mistralai/voxtral-small-24b-2507-stt":{"mode":"chat","base_model":"voxtral-small-24b-2507-stt","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/mistralai/voxtral-mini-3b-2507":{"mode":"chat","base_model":"voxtral-mini-3b-2507","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/qwen/qwen3-asr-1.7b":{"mode":"chat","base_model":"qwen3-asr-1.7b","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/qwen/qwen3-asr-0.6b":{"mode":"chat","base_model":"qwen3-asr-0.6b","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/openai/gpt-transcribe":{"mode":"chat","base_model":"gpt-transcribe","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/fish-audio/transcribe-1":{"mode":"chat","base_model":"transcribe-1","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/fish-audio/s1":{"mode":"chat","base_model":"s1","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/fish-audio/s2-pro":{"mode":"chat","base_model":"s2-pro","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/fish-audio/s2.1-pro":{"mode":"chat","base_model":"s2.1-pro","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/microsoft/mai-voice-2-flash":{"mode":"chat","base_model":"mai-voice-2-flash","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/qwen/qwen-audio-3.0-tts-flash":{"mode":"chat","base_model":"qwen-audio-3.0-tts-flash","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/qwen/qwen-audio-3.0-tts-plus":{"mode":"chat","base_model":"qwen-audio-3.0-tts-plus","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/x-ai/grok-stt-1.0":{"mode":"chat","base_model":"grok-stt-1.0","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/openrouter/auto-beta":{"mode":"chat","base_model":"auto-beta","provider":"openrouter","max_input_tokens":2000000,"max_tokens":2000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_vision":true,"supports_audio_input":true,"supports_pdf_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":2000000}}]},"openrouter/deepgram/aura-2":{"mode":"chat","base_model":"aura-2","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/minimax/speech-2.8-hd":{"mode":"chat","base_model":"speech-2.8-hd","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/minimax/speech-2.8-turbo":{"mode":"chat","base_model":"speech-2.8-turbo","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/deepgram/nova-3":{"mode":"chat","base_model":"nova-3","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/openrouter/fusion":{"mode":"chat","base_model":"fusion","provider":"openrouter","max_input_tokens":1000000,"max_tokens":1000000,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"openrouter/microsoft/mai-voice-2":{"mode":"chat","base_model":"mai-voice-2","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/microsoft/mai-transcribe-1.5":{"mode":"chat","base_model":"mai-transcribe-1.5","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/nvidia/parakeet-tdt-0.6b-v3":{"mode":"chat","base_model":"parakeet-tdt-0.6b-v3","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/mistralai/voxtral-mini-transcribe":{"mode":"chat","base_model":"voxtral-mini-transcribe","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/x-ai/grok-voice-tts-1.0":{"mode":"chat","base_model":"grok-voice-tts-1.0","provider":"openrouter","max_input_tokens":15000,"max_output_tokens":13500,"max_tokens":13500,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":13500}}]},"openrouter/qwen/qwen3-asr-flash-2026-02-10":{"mode":"chat","base_model":"qwen3-asr-flash-2026-02-10","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/google/chirp-3":{"mode":"chat","base_model":"chirp-3","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/openai/gpt-4o-mini-transcribe":{"mode":"chat","base_model":"gpt-4o-mini-transcribe","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":115200,"max_tokens":115200,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":115200}}]},"openrouter/openai/whisper-large-v3":{"mode":"chat","base_model":"whisper-large-v3","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/openai/whisper-large-v3-turbo":{"mode":"chat","base_model":"whisper-large-v3-turbo","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/openai/whisper-1":{"mode":"chat","base_model":"whisper-1","provider":"openrouter","max_input_tokens":0,"max_output_tokens":0,"max_tokens":0,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"openrouter/openai/gpt-4o-transcribe":{"mode":"chat","base_model":"gpt-4o-transcribe","provider":"openrouter","max_input_tokens":128000,"max_output_tokens":115200,"max_tokens":115200,"supports_audio_input":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":115200}}]},"openrouter/google/gemini-3.1-flash-tts-preview":{"mode":"chat","base_model":"gemini-3.1-flash-tts-preview","provider":"openrouter","max_input_tokens":32768,"max_output_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"openrouter/canopylabs/orpheus-3b-0.1-ft":{"mode":"chat","base_model":"orpheus-3b-0.1-ft","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":3686,"max_tokens":3686,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":3686,"range":{"min":1,"max":3686}}]},"openrouter/sesame/csm-1b":{"mode":"chat","base_model":"csm-1b","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":3686,"max_tokens":3686,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":3686,"range":{"min":1,"max":3686}}]},"openrouter/hexgrad/kokoro-82m":{"mode":"chat","base_model":"kokoro-82m","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":3686,"max_tokens":3686,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":3686,"range":{"min":1,"max":3686}}]},"openrouter/openrouter/pareto-code":{"mode":"chat","base_model":"pareto-code","provider":"openrouter","max_input_tokens":2000000,"max_tokens":2000000,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":2000000}}]},"openrouter/mistralai/voxtral-mini-tts-2603":{"mode":"chat","base_model":"voxtral-mini-tts-2603","provider":"openrouter","max_input_tokens":4096,"max_output_tokens":3276,"max_tokens":3276,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":3276,"range":{"min":1,"max":3276}}]},"openrouter/openai/gpt-oss-20b:batch":{"mode":"chat","base_model":"gpt-oss-20b:batch","provider":"openrouter","max_input_tokens":131072,"max_output_tokens":117964,"max_tokens":117964,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":117964}}]},"vertex_ai/xai/grok-4.20-beta-0309-non-reasoning":{"mode":"chat","base_model":"grok-4.20","max_input_tokens":2000000,"max_output_tokens":2000000,"max_tokens":2000000,"source":"https://docs.x.ai/docs/models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"supports_web_search":true,"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":2000000}}]},"bedrock/moonshotai.kimi-k2.6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"source":"https://platform.kimi.ai/docs/guide/kimi-k2-6-quickstart","supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_video_input":true,"supports_vision":true,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"moonshotai.kimi-k2.6":{"mode":"chat","base_model":"kimi-k2.6","max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"source":"https://aws.amazon.com/bedrock/pricing/","provider":"moonshot","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"vertex_ai/gemma-4-31b":{"mode":"chat","base_model":"gemma-4","max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"tpm":250000,"rpm":10,"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"vertex_ai/gemma-4-26b-a4b-it-maas":{"mode":"chat","base_model":"gemma-4","max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"tpm":250000,"rpm":10,"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"azure/DeepSeek-V4-Flash":{"mode":"chat","base_model":"deepseek-v4-flash","provider":"azure","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"azure/DeepSeek-V4-Pro":{"mode":"chat","base_model":"deepseek-v4-pro","provider":"azure","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_prompt_caching":true,"model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"moonshot/kimi-k2.7-code-highspeed":{"mode":"chat","base_model":"kimi-k2.7-code-highspeed","max_input_tokens":262144,"max_tokens":262144,"supported_endpoints":["/v1/chat/completions"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_reasoning":true,"supports_prompt_caching":true,"supports_vision":true,"provider":"moonshot","model":"kimi-k2.7-code-highspeed","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"deepseek/deepseek-v4-pro-0813":{"mode":"chat","base_model":"deepseek-v4-pro","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"deepseek","peak_hours":{"timezone":"UTC","windows":[{"days":[1,2,3,4,5],"start":"01:00","end":"04:00"},{"days":[1,2,3,4,5],"start":"06:00","end":"10:00"}]},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"deepseek/deepSeek-v4-flash-0731":{"mode":"chat","base_model":"deepseek-v4-flash","max_input_tokens":1000000,"max_output_tokens":393216,"max_tokens":393216,"source":"https://api-docs.deepseek.com/quick_start/pricing","supported_endpoints":["/v1/chat/completions"],"supports_assistant_prefill":true,"supports_function_calling":true,"supports_native_streaming":true,"supports_parallel_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":false,"provider":"deepseek","peak_hours":{"timezone":"UTC","windows":[{"days":[1,2,3,4,5],"start":"01:00","end":"04:00"},{"days":[1,2,3,4,5],"start":"06:00","end":"10:00"}]},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":393216}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-oss-safeguard-120b":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-oss-safeguard-20b":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"supported_endpoints":["/v1/chat/completions"],"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"bedrock_mantle/us-gov-east-1/openai.gpt-5.6-terra":{"mode":"responses","base_model":"gpt-5.6-terra","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-east-1/openai.gpt-5.6-luna":{"mode":"responses","base_model":"gpt-5.6-luna","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_response_schema":false,"supports_computer_use":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_response_schema":false,"supports_computer_use":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_response_schema":false,"supports_computer_use":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_response_schema":true,"supports_computer_use":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"bedrock_mantle/us-gov-west-1/anthropic.claude-3-7-sonnet-20250219-v1:0":{"mode":"chat","base_model":"claude-3-7-sonnet","provider":"bedrock_mantle","supports_prompt_caching":true,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"bedrock_mantle/us-gov-west-1/anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","provider":"bedrock_mantle","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"bedrock_mantle/us-gov-west-1/anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","ca-central-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"],"max_input_tokens":200000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"bedrock_mantle/us-gov-east-1/anthropic.claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_response_schema":false,"supports_computer_use":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-east-1/anthropic.claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_response_schema":false,"supports_computer_use":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-east-1/anthropic.claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_response_schema":false,"supports_computer_use":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-east-1/anthropic.claude-sonnet-4-5-20250929-v1:0":{"mode":"chat","base_model":"claude-sonnet-4-5","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_response_schema":true,"supports_computer_use":true,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"bedrock_mantle/us-gov-east-1/anthropic.claude-3-7-sonnet-20250219-v1:0":{"mode":"chat","base_model":"claude-3-7-sonnet","provider":"bedrock_mantle","supports_prompt_caching":true,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"bedrock_mantle/us-gov-east-1/anthropic.claude-3-5-sonnet-20240620-v1:0":{"mode":"chat","base_model":"claude-3-5-sonnet","provider":"bedrock_mantle","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"bedrock_mantle/us-gov-east-1/anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-2","ca-central-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-west-1","eu-west-2","eu-west-3","ap-northeast-1","ap-northeast-2","ap-south-1","ap-southeast-1","ap-southeast-2","sa-east-1"],"max_input_tokens":200000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-oss-20b-1:0":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_response_schema":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-oss-120b-1:0":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_response_schema":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock_mantle/us-gov-east-1/openai.gpt-oss-20b-1:0":{"mode":"chat","base_model":"gpt-oss-20b","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_response_schema":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock_mantle/us-gov-east-1/openai.gpt-oss-120b-1:0":{"mode":"chat","base_model":"gpt-oss-120b","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_response_schema":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"bedrock_mantle/us-gov-east-1/nvidia.nemotron-nano-9b-v2":{"mode":"chat","base_model":"nemotron-nano-9b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/us-gov-east-1/nvidia.nemotron-nano-12b-v2":{"mode":"chat","base_model":"nemotron-nano-12b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/us-gov-east-1/nvidia.nemotron-nano-3-30b":{"mode":"chat","base_model":"nemotron-nano-3-30b","max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/us-gov-east-1/nvidia.nemotron-super-3-120b":{"mode":"chat","base_model":"nemotron-super-3-120b","max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-southeast-2","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock_mantle/us-gov-west-1/nvidia.nemotron-nano-9b-v2":{"mode":"chat","base_model":"nemotron-nano-9b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/us-gov-west-1/nvidia.nemotron-nano-12b-v2":{"mode":"chat","base_model":"nemotron-nano-12b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/us-gov-west-1/nvidia.nemotron-nano-3-30b":{"mode":"chat","base_model":"nemotron-nano-3-30b","max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/us-gov-west-1/nvidia.nemotron-super-3-120b":{"mode":"chat","base_model":"nemotron-super-3-120b","max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-southeast-2","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock_mantle/nvidia.nemotron-nano-9b-v2":{"mode":"chat","base_model":"nemotron-nano-9b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/nvidia.nemotron-nano-12b-v2":{"mode":"chat","base_model":"nemotron-nano-12b-v2","max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"supports_vision":true,"provider":"bedrock_mantle","supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/nvidia.nemotron-nano-3-30b":{"mode":"chat","base_model":"nemotron-nano-3-30b","max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","us-gov-west-1","eu-central-1","eu-north-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/nvidia.nemotron-super-3-120b":{"mode":"chat","base_model":"nemotron-super-3-120b","max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"supported_endpoints":["/v1/chat/completions","/v1/batch"],"provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-southeast-2","sa-east-1","ap-southeast-4","us-gov-west-1","us-gov-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock_mantle/anthropic.claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_adaptive_thinking":true,"thinking_always_on":true,"supports_mid_conversation_system":true,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_forced_tool_use":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":true,"supports_max_reasoning_effort":true,"supports_output_config":true,"bedrock_output_config_effort_ceiling":"xhigh","supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"provider":"bedrock_mantle","supports_forced_tool_choice":false,"supports_web_search":true,"tool_use_system_prompt_tokens":346,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1","us-gov-east-1","us-gov-west-1"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1000000,"max_tokens":1000000,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_tool_choice":true,"supports_reasoning":true,"supports_response_schema":true,"supports_prompt_caching":true,"supports_web_search":true,"provider":"bedrock","model":"kimi-k3","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"min_output_tokens":16,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"bedrock/global.moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"provider":"bedrock","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_video_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_reasoning":true,"supports_native_streaming":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"source":"https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-moonshot-ai-kimi-k3.html","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"bedrock/us.moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","max_input_tokens":1000000,"max_output_tokens":1048576,"max_tokens":1000000,"provider":"bedrock","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_video_input":false,"supports_function_calling":true,"supports_tool_choice":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_reasoning":true,"supports_native_streaming":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"source":"https://aws.amazon.com/bedrock/pricing/","min_output_tokens":16,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"us-gov.anthropic.claude-3-haiku-20240307-v1:0":{"mode":"chat","base_model":"claude-3-haiku","deprecation_date":"2026-09-10","max_input_tokens":200000,"max_output_tokens":4096,"max_tokens":4096,"supports_cache_point":false,"supports_function_calling":true,"supports_pdf_input":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock","source":"https://aws.amazon.com/bedrock/pricing/","is_deprecated":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":4096}}]},"bedrock_mantle/anthropic.claude-opus-5-5":{"mode":"chat","base_model":"anthropic.claude-opus-5-5","provider":"bedrock_mantle","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"prompt_cache_min_tokens":512,"default_reasoning_effort":"medium","thinking_always_on":true,"supports_reasoning":true,"supports_adaptive_thinking":true,"supports_vision":true,"supports_image_input":true,"supports_pdf_input":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_tool_choice":true,"supports_forced_tool_choice":false,"supports_forced_tool_use":false,"supports_system_messages":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_output_config":true,"supports_assistant_prefill":false,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/global.openai.gpt-6-sol":{"mode":"chat","base_model":"gpt-6-sol","provider":"bedrock","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_endpoint_uplift_multiplier":1.1,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"default_reasoning_effort":"medium","reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_function_calling":true,"supports_tool_choice":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_image_input":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_none_reasoning_effort":true,"supports_low_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"supports_prompt_caching":true,"supports_prompt_cache_breakpoint":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_native_streaming":true,"supports_system_messages":true,"supports_web_search":true,"supports_file_search":true,"supports_computer_use":true,"supports_tool_search":true,"supports_service_tier":true,"source":"https://developers.openai.com/api/docs/models/gpt-6-sol","metadata":{"notes":"For prompts over 272K input tokens, OpenAI charges 2x input/cache and 1.5x output for the full request. Batch and Flex are 50% of Standard. Fast is 2x the applicable rate. The current Bifrost schema has no dedicated Fast-above-272K pricing fields, so the Fast fields here represent short-context Fast pricing."},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/global.openai.gpt-6-sol":{"mode":"chat","base_model":"gpt-6-sol","provider":"bedrock_mantle","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_endpoint_uplift_multiplier":1.1,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"default_reasoning_effort":"medium","reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_function_calling":true,"supports_tool_choice":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_image_input":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_none_reasoning_effort":true,"supports_low_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"supports_prompt_caching":true,"supports_prompt_cache_breakpoint":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_native_streaming":true,"supports_system_messages":true,"supports_web_search":true,"supports_file_search":true,"supports_computer_use":true,"supports_tool_search":true,"supports_service_tier":true,"source":"https://developers.openai.com/api/docs/models/gpt-6-sol","metadata":{"notes":"For prompts over 272K input tokens, OpenAI charges 2x input/cache and 1.5x output for the full request. Batch and Flex are 50% of Standard. Fast is 2x the applicable rate. The current Bifrost schema has no dedicated Fast-above-272K pricing fields, so the Fast fields here represent short-context Fast pricing."},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/global.openai.gpt-6-luna":{"mode":"chat","base_model":"gpt-6-luna","provider":"bedrock","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_endpoint_uplift_multiplier":1.1,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"default_reasoning_effort":"medium","reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_function_calling":true,"supports_tool_choice":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_image_input":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_none_reasoning_effort":true,"supports_low_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"supports_prompt_caching":true,"supports_prompt_cache_breakpoint":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_native_streaming":true,"supports_system_messages":true,"supports_web_search":true,"supports_file_search":true,"supports_computer_use":true,"supports_tool_search":true,"supports_service_tier":true,"source":"https://developers.openai.com/api/docs/models/gpt-6-luna","metadata":{"notes":"For prompts over 272K input tokens, OpenAI charges 2x input/cache and 1.5x output for the full request. Batch and Flex are 50% of Standard. Fast is 2x the applicable rate. The current Bifrost schema has no dedicated Fast-above-272K pricing fields, so the Fast fields here represent short-context Fast pricing."},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/global.openai.gpt-6-luna":{"mode":"chat","base_model":"gpt-6-luna","provider":"bedrock_mantle","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_endpoint_uplift_multiplier":1.1,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"default_reasoning_effort":"medium","reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_function_calling":true,"supports_tool_choice":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_image_input":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_none_reasoning_effort":true,"supports_low_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"supports_prompt_caching":true,"supports_prompt_cache_breakpoint":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_native_streaming":true,"supports_system_messages":true,"supports_web_search":true,"supports_file_search":true,"supports_computer_use":true,"supports_tool_search":true,"supports_service_tier":true,"source":"https://developers.openai.com/api/docs/models/gpt-6-luna","metadata":{"notes":"For prompts over 272K input tokens, OpenAI charges 2x input/cache and 1.5x output for the full request. Batch and Flex are 50% of Standard. Fast is 2x the applicable rate. The current Bifrost schema has no dedicated Fast-above-272K pricing fields, so the Fast fields here represent short-context Fast pricing."},"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/us.anthropic.claude-opus-5-5":{"mode":"chat","base_model":"claude-opus-5-5","bedrock_converse_supports_strict_tools":false,"supports_adaptive_thinking":true,"supports_mid_conversation_system":true,"supports_tool_search":true,"max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"supports_assistant_prefill":false,"supports_computer_use":true,"supports_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_sampling_params":false,"supports_tool_choice":true,"supports_vision":true,"supports_xhigh_reasoning_effort":true,"supports_native_structured_output":false,"supports_max_reasoning_effort":true,"supports_output_config":true,"supports_parallel_tool_use_config":true,"prompt_cache_min_tokens":512,"source":"https://aws.amazon.com/bedrock/pricing/","thinking_always_on":true,"supports_forced_tool_use":false,"provider":"bedrock","supports_audio_input":false,"supports_audio_output":false,"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","us-gov-west-1","us-gov-east-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1","mx-central-1"],"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/openai.gpt-6-luna":{"mode":"chat","base_model":"gpt-6-luna","provider":"bedrock_mantle","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_endpoint_uplift_multiplier":1.1,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"default_reasoning_effort":"medium","reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_function_calling":true,"supports_tool_choice":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_image_input":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_none_reasoning_effort":true,"supports_low_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"supports_prompt_caching":true,"supports_prompt_cache_breakpoint":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_native_streaming":true,"supports_system_messages":true,"supports_web_search":true,"supports_file_search":true,"supports_computer_use":true,"supports_tool_search":true,"supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/openai.gpt-6-luna":{"mode":"chat","base_model":"gpt-6-luna","provider":"bedrock","max_input_tokens":1050000,"max_output_tokens":128000,"max_tokens":128000,"regional_endpoint_uplift_multiplier":1.1,"supported_endpoints":["/v1/chat/completions","/v1/responses","/v1/batch"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"default_reasoning_effort":"medium","reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"supports_function_calling":true,"supports_tool_choice":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_image_input":true,"supports_reasoning":true,"supports_reasoning_with_tool_calls":true,"supports_none_reasoning_effort":true,"supports_low_reasoning_effort":true,"supports_xhigh_reasoning_effort":true,"supports_max_reasoning_effort":true,"supports_prompt_caching":true,"supports_prompt_cache_breakpoint":true,"supports_response_schema":true,"supports_native_structured_output":true,"supports_native_streaming":true,"supports_system_messages":true,"supports_web_search":true,"supports_file_search":true,"supports_computer_use":true,"supports_tool_search":true,"supports_service_tier":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"gemini/gemini-robotics-er-1.6-preview":{"mode":"chat","base_model":"gemini-robotics-er-1.6-preview","provider":"gemini","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":1064000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"gemini/gemini-3.1-flash-lite-preview":{"mode":"chat","base_model":"gemini-3.1-flash-lite-preview","provider":"gemini","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":1064000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"gemini/deep-research-preview-04-2026":{"mode":"chat","base_model":"deep-research-preview-04-2026","provider":"gemini","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":1064000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"gemini/deep-research-max-preview-04-2026":{"mode":"chat","base_model":"deep-research-max-preview-04-2026","provider":"gemini","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":1064000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"gemini/antigravity-preview-05-2026":{"mode":"chat","base_model":"antigravity-preview-05-2026","provider":"gemini","max_input_tokens":1000000,"max_output_tokens":64000,"max_tokens":1064000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"opencode-zen/grok-4.6":{"mode":"chat","base_model":"grok-4.6","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/glm-5.3-flash":{"mode":"chat","base_model":"glm-5.3-flash","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/glm-5.3":{"mode":"chat","base_model":"glm-5.3","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/glm-5.2":{"mode":"chat","base_model":"glm-5.2","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/glm-5.1":{"mode":"chat","base_model":"glm-5.1","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/kimi-k3":{"mode":"chat","base_model":"kimi-k3","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/kimi-k2.7-code":{"mode":"chat","base_model":"kimi-k2.7-code","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/kimi-k2.6":{"mode":"chat","base_model":"kimi-k2.6","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/longcat-2.0":{"mode":"chat","base_model":"longcat-2.0","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/deepseek-v4.1-flash":{"mode":"chat","base_model":"deepseek-v4.1-flash","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/deepseek-v4-pro":{"mode":"chat","base_model":"deepseek-v4-pro","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/deepseek-v4-flash":{"mode":"chat","base_model":"deepseek-v4-flash","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/deepseek-v4-flash-vision-exp":{"mode":"chat","base_model":"deepseek-v4-flash-vision-exp","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/mimo-v2.5":{"mode":"chat","base_model":"mimo-v2.5","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/mimo-v2.5-pro":{"mode":"chat","base_model":"mimo-v2.5-pro","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/minimax-m3":{"mode":"chat","base_model":"minimax-m3","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/minimax-m2.7":{"mode":"chat","base_model":"minimax-m2.7","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/muse-spark-1.3-contributor":{"mode":"chat","base_model":"muse-spark-1.3-contributor","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/muse-spark-1.2-contributor":{"mode":"chat","base_model":"muse-spark-1.2-contributor","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/qwen3.8-max":{"mode":"chat","base_model":"qwen3.8-max","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/qwen3.8-flash":{"mode":"chat","base_model":"qwen3.8-flash","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/qwen3.7-max":{"mode":"chat","base_model":"qwen3.7-max","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/qwen3.7-plus":{"mode":"chat","base_model":"qwen3.7-plus","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-go/qwen3.6-plus":{"mode":"chat","base_model":"qwen3.6-plus","provider":"opencode-go","source":"https://opencode.ai/docs/go/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/hy4-preview":{"mode":"chat","base_model":"hy4-preview","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/hy3":{"mode":"chat","base_model":"hy3","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-6-astra":{"mode":"chat","base_model":"gpt-6-astra","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.6-sol":{"mode":"chat","base_model":"gpt-5.6-sol","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.5":{"mode":"chat","base_model":"gpt-5.5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.5-pro":{"mode":"chat","base_model":"gpt-5.5-pro","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.4":{"mode":"chat","base_model":"gpt-5.4","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.4-pro":{"mode":"chat","base_model":"gpt-5.4-pro","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.4-mini":{"mode":"chat","base_model":"gpt-5.4-mini","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.4-nano":{"mode":"chat","base_model":"gpt-5.4-nano","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.3-codex":{"mode":"chat","base_model":"gpt-5.3-codex","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.3-codex-spark":{"mode":"chat","base_model":"gpt-5.3-codex-spark","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.2":{"mode":"chat","base_model":"gpt-5.2","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.2-codex":{"mode":"chat","base_model":"gpt-5.2-codex","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.1":{"mode":"chat","base_model":"gpt-5.1","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.1-codex":{"mode":"chat","base_model":"gpt-5.1-codex","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.1-codex-max":{"mode":"chat","base_model":"gpt-5.1-codex-max","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5.1-codex-mini":{"mode":"chat","base_model":"gpt-5.1-codex-mini","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5":{"mode":"chat","base_model":"gpt-5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5-codex":{"mode":"chat","base_model":"gpt-5-codex","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gpt-5-nano":{"mode":"chat","base_model":"gpt-5-nano","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-fable-5-1":{"mode":"chat","base_model":"claude-fable-5-1","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-fable-5":{"mode":"chat","base_model":"claude-fable-5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-opus-5":{"mode":"chat","base_model":"claude-opus-5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-opus-4-8":{"mode":"chat","base_model":"claude-opus-4-8","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-opus-4-7":{"mode":"chat","base_model":"claude-opus-4-7","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-opus-4-6":{"mode":"chat","base_model":"claude-opus-4-6","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-opus-4-5":{"mode":"chat","base_model":"claude-opus-4-5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-sonnet-5":{"mode":"chat","base_model":"claude-sonnet-5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-sonnet-4-6":{"mode":"chat","base_model":"claude-sonnet-4-6","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-sonnet-4-5":{"mode":"chat","base_model":"claude-sonnet-4-5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gemini-3.8-flash":{"mode":"chat","base_model":"gemini-3.8-flash","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gemini-3.7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gemini-3.6-flash":{"mode":"chat","base_model":"gemini-3.6-flash","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gemini-3.5-flash":{"mode":"chat","base_model":"gemini-3.5-flash","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gemini-3.5-flash-lite":{"mode":"chat","base_model":"gemini-3.5-flash-lite","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gemini-3.1-pro":{"mode":"chat","base_model":"gemini-3.1-pro","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/gemini-3-flash":{"mode":"chat","base_model":"gemini-3-flash","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/grok-4.5":{"mode":"chat","base_model":"grok-4.5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/grok-build-0.1":{"mode":"chat","base_model":"grok-build-0.1","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/muse-spark-1.3":{"mode":"chat","base_model":"muse-spark-1.3","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/muse-spark-1.2":{"mode":"chat","base_model":"muse-spark-1.2","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/qwen3.5-plus":{"mode":"chat","base_model":"qwen3.5-plus","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/glm-5":{"mode":"chat","base_model":"glm-5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/kimi-k2.5":{"mode":"chat","base_model":"kimi-k2.5","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/big-pickle":{"mode":"chat","base_model":"big-pickle","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/mimo-v2.5-free":{"mode":"chat","base_model":"mimo-v2.5-free","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/ling-3.0-flash-fin-free":{"mode":"chat","base_model":"ling-3.0-flash-fin-free","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/nemotron-3-ultra-free":{"mode":"chat","base_model":"nemotron-3-ultra-free","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/nemotron-3.5-lightning-free":{"mode":"chat","base_model":"nemotron-3.5-lightning-free","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"opencode-zen/muse-spark-1.3-contributor-free":{"mode":"chat","base_model":"muse-spark-1.3-contributor-free","provider":"opencode-zen","source":"https://opencode.ai/docs/zen/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/novita/deepseek-ai/DeepSeek-V4.1-Flash":{"mode":"chat","base_model":"DeepSeek-V4.1-Flash","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/baseten/deepseek-ai/DeepSeek-V4.1-Flash":{"mode":"chat","base_model":"DeepSeek-V4.1-Flash","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V4.1-Flash":{"mode":"chat","base_model":"DeepSeek-V4.1-Flash","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/Qwen/Qwen3.8-27B":{"mode":"chat","base_model":"Qwen3.8-27B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.8-27B","max_input_tokens":1000000,"max_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"huggingface/cerebras/Qwen/Qwen3.8-27B":{"mode":"chat","base_model":"Qwen3.8-27B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.8-27B","max_input_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"huggingface/deepinfra/Qwen/Qwen3.8-27B":{"mode":"chat","base_model":"Qwen3.8-27B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.8-27B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"GLM-5.3-Flash","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.3-Flash","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/together/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"GLM-5.3-Flash","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.3-Flash","max_input_tokens":1048575,"max_tokens":1048575,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048575}}]},"huggingface/baseten/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"GLM-5.3-Flash","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.3-Flash","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/zai-org/GLM-5.3-Flash":{"mode":"chat","base_model":"GLM-5.3-Flash","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.3-Flash","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/meta-llama/Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"Llama-3.1-8B-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct","max_input_tokens":16384,"max_tokens":16384,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"huggingface/nscale/meta-llama/Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"Llama-3.1-8B-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/meta-llama/Llama-3.1-8B-Instruct":{"mode":"chat","base_model":"Llama-3.1-8B-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/together/moonshotai/Kimi-K3":{"mode":"chat","base_model":"Kimi-K3","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K3","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/fireworks-ai/moonshotai/Kimi-K3":{"mode":"chat","base_model":"Kimi-K3","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K3","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/baseten/moonshotai/Kimi-K3":{"mode":"chat","base_model":"Kimi-K3","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K3","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/moonshotai/Kimi-K3":{"mode":"chat","base_model":"Kimi-K3","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K3","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/zai-org/GLM-5.3":{"mode":"chat","base_model":"GLM-5.3","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.3","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/together/zai-org/GLM-5.3":{"mode":"chat","base_model":"GLM-5.3","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.3","max_input_tokens":1048575,"max_tokens":1048575,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048575}}]},"huggingface/baseten/zai-org/GLM-5.3":{"mode":"chat","base_model":"GLM-5.3","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.3","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/zai-org/GLM-5.3":{"mode":"chat","base_model":"GLM-5.3","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.3","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/google/gemma-4-31B-it":{"mode":"chat","base_model":"gemma-4-31B-it","provider":"huggingface","source":"https://huggingface.co/google/gemma-4-31B-it","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/google/gemma-4-31B-it":{"mode":"chat","base_model":"gemma-4-31B-it","provider":"huggingface","source":"https://huggingface.co/google/gemma-4-31B-it","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/together/Qwen/Qwen3.5-9B":{"mode":"chat","base_model":"Qwen3.5-9B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-9B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/ovhcloud/Qwen/Qwen3.5-9B":{"mode":"chat","base_model":"Qwen3.5-9B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-9B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/Qwen/Qwen3.5-9B":{"mode":"chat","base_model":"Qwen3.5-9B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-9B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/scaleway/Qwen/Qwen3.6-35B-A3B":{"mode":"chat","base_model":"Qwen3.6-35B-A3B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/deepinfra/Qwen/Qwen3.6-35B-A3B":{"mode":"chat","base_model":"Qwen3.6-35B-A3B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.6-35B-A3B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/groq/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/cerebras/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/nscale/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/together/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/fireworks-ai/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/scaleway/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/baseten/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":128072,"max_tokens":128072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128072}}]},"huggingface/ovhcloud/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/openai/gpt-oss-120b":{"mode":"chat","base_model":"gpt-oss-120b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-120b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/nscale/Qwen/Qwen3-8B":{"mode":"chat","base_model":"Qwen3-8B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-8B","max_input_tokens":40960,"max_tokens":40960,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"huggingface/together/meta-models/Muse-Glimmer-30B":{"mode":"chat","base_model":"Muse-Glimmer-30B","provider":"huggingface","source":"https://huggingface.co/meta-models/Muse-Glimmer-30B","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/meta-models/Muse-Glimmer-30B":{"mode":"chat","base_model":"Muse-Glimmer-30B","provider":"huggingface","source":"https://huggingface.co/meta-models/Muse-Glimmer-30B","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/deepseek-ai/DeepSeek-R1":{"mode":"chat","base_model":"DeepSeek-R1","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-R1","max_input_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"huggingface/novita/Qwen/Qwen3.8-2.4T-A95B":{"mode":"chat","base_model":"Qwen3.8-2.4T-A95B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","max_input_tokens":1000000,"max_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"huggingface/together/Qwen/Qwen3.8-2.4T-A95B":{"mode":"chat","base_model":"Qwen3.8-2.4T-A95B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","max_input_tokens":1010000,"max_tokens":1010000,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1010000}}]},"huggingface/deepinfra/Qwen/Qwen3.8-2.4T-A95B":{"mode":"chat","base_model":"Qwen3.8-2.4T-A95B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/groq/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-20b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-20b","max_input_tokens":131072,"max_tokens":131072,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/nscale/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-20b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/ovhcloud/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-20b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/openai/gpt-oss-20b":{"mode":"chat","base_model":"gpt-oss-20b","provider":"huggingface","source":"https://huggingface.co/openai/gpt-oss-20b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"DeepSeek-V4-Flash-0731","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/together/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"DeepSeek-V4-Flash-0731","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/baseten/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"DeepSeek-V4-Flash-0731","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731":{"mode":"chat","base_model":"DeepSeek-V4-Flash-0731","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/Qwen/Qwen3-Coder-Next":{"mode":"chat","base_model":"Qwen3-Coder-Next","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-Coder-Next","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/google/gemma-4-26B-A4B-it":{"mode":"chat","base_model":"gemma-4-26B-A4B-it","provider":"huggingface","source":"https://huggingface.co/google/gemma-4-26B-A4B-it","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/scaleway/google/gemma-4-26B-A4B-it":{"mode":"chat","base_model":"gemma-4-26B-A4B-it","provider":"huggingface","source":"https://huggingface.co/google/gemma-4-26B-A4B-it","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/deepinfra/google/gemma-4-26B-A4B-it":{"mode":"chat","base_model":"gemma-4-26B-A4B-it","provider":"huggingface","source":"https://huggingface.co/google/gemma-4-26B-A4B-it","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/deepseek-ai/DeepSeek-V4-Pro":{"mode":"chat","base_model":"DeepSeek-V4-Pro","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/baseten/deepseek-ai/DeepSeek-V4-Pro":{"mode":"chat","base_model":"DeepSeek-V4-Pro","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V4-Pro":{"mode":"chat","base_model":"DeepSeek-V4-Pro","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/deepseek-ai/DeepSeek-V4-Flash":{"mode":"chat","base_model":"DeepSeek-V4-Flash","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V4-Flash":{"mode":"chat","base_model":"DeepSeek-V4-Flash","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/tencent/Hy4-preview":{"mode":"chat","base_model":"Hy4-preview","provider":"huggingface","source":"https://huggingface.co/tencent/Hy4-preview","max_input_tokens":1000000,"max_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"huggingface/deepinfra/Qwen/Qwen3.6-27B":{"mode":"chat","base_model":"Qwen3.6-27B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.6-27B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"mode":"chat","base_model":"DeepSeek-V4-Flash-Vision-Exp","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"mode":"chat","base_model":"DeepSeek-V4-Flash-Vision-Exp","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/inclusionAI/Ling-3.0-flash-Fin":{"mode":"chat","base_model":"Ling-3.0-flash-Fin","provider":"huggingface","source":"https://huggingface.co/inclusionAI/Ling-3.0-flash-Fin","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/MiniMaxAI/MiniMax-M3":{"mode":"chat","base_model":"MiniMax-M3","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M3","max_input_tokens":1000000,"max_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"huggingface/together/MiniMaxAI/MiniMax-M3":{"mode":"chat","base_model":"MiniMax-M3","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M3","max_input_tokens":524288,"max_tokens":524288,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"huggingface/fireworks-ai/MiniMaxAI/MiniMax-M3":{"mode":"chat","base_model":"MiniMax-M3","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M3","max_input_tokens":512000,"max_tokens":512000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":512000}}]},"huggingface/deepinfra/MiniMaxAI/MiniMax-M3":{"mode":"chat","base_model":"MiniMax-M3","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M3","max_input_tokens":524288,"max_tokens":524288,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"huggingface/nscale/Qwen/Qwen3-4B-Instruct-2507":{"mode":"chat","base_model":"Qwen3-4B-Instruct-2507","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-4B-Instruct-2507","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"Llama-3.3-70B-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct","max_input_tokens":12288,"max_tokens":12288,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":12288}}]},"huggingface/together/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"Llama-3.3-70B-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/scaleway/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"Llama-3.3-70B-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/ovhcloud/meta-llama/Llama-3.3-70B-Instruct":{"mode":"chat","base_model":"Llama-3.3-70B-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/google/gemma-3-4b-it":{"mode":"chat","base_model":"gemma-3-4b-it","provider":"huggingface","source":"https://huggingface.co/google/gemma-3-4b-it","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/inclusionAI/Ling-3.0-flash-VL":{"mode":"chat","base_model":"Ling-3.0-flash-VL","provider":"huggingface","source":"https://huggingface.co/inclusionAI/Ling-3.0-flash-VL","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/scaleway/Qwen/Qwen3-Coder-30B-A3B-Instruct":{"mode":"chat","base_model":"Qwen3-Coder-30B-A3B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/baseten/thinkingmachines/Inkling-Small":{"mode":"chat","base_model":"Inkling-Small","provider":"huggingface","source":"https://huggingface.co/thinkingmachines/Inkling-Small","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/thinkingmachines/Inkling-Small":{"mode":"chat","base_model":"Inkling-Small","provider":"huggingface","source":"https://huggingface.co/thinkingmachines/Inkling-Small","max_input_tokens":524288,"max_tokens":524288,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"huggingface/novita/zai-org/GLM-5.2":{"mode":"chat","base_model":"GLM-5.2","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.2","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/together/zai-org/GLM-5.2":{"mode":"chat","base_model":"GLM-5.2","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.2","max_input_tokens":1048575,"max_tokens":1048575,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048575}}]},"huggingface/fireworks-ai/zai-org/GLM-5.2":{"mode":"chat","base_model":"GLM-5.2","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.2","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/scaleway/zai-org/GLM-5.2":{"mode":"chat","base_model":"GLM-5.2","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.2","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/baseten/zai-org/GLM-5.2":{"mode":"chat","base_model":"GLM-5.2","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.2","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/zai-org/GLM-5.2":{"mode":"chat","base_model":"GLM-5.2","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.2","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/together/thinkingmachines/Inkling":{"mode":"chat","base_model":"Inkling","provider":"huggingface","source":"https://huggingface.co/thinkingmachines/Inkling","max_input_tokens":524288,"max_tokens":524288,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"huggingface/baseten/thinkingmachines/Inkling":{"mode":"chat","base_model":"Inkling","provider":"huggingface","source":"https://huggingface.co/thinkingmachines/Inkling","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/thinkingmachines/Inkling":{"mode":"chat","base_model":"Inkling","provider":"huggingface","source":"https://huggingface.co/thinkingmachines/Inkling","max_input_tokens":524288,"max_tokens":524288,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":524288}}]},"huggingface/publicai/swiss-ai/Apertus-v1.5-70B":{"mode":"chat","base_model":"Apertus-v1.5-70B","provider":"huggingface","source":"https://huggingface.co/swiss-ai/Apertus-v1.5-70B","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/novita/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"DeepSeek-V4-Pro-0813","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/together/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"DeepSeek-V4-Pro-0813","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/baseten/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"DeepSeek-V4-Pro-0813","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813":{"mode":"chat","base_model":"DeepSeek-V4-Pro-0813","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/XiaomiMiMo/MiMo-V2.5-Pro":{"mode":"chat","base_model":"MiMo-V2.5-Pro","provider":"huggingface","source":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/XiaomiMiMo/MiMo-V2.5-Pro":{"mode":"chat","base_model":"MiMo-V2.5-Pro","provider":"huggingface","source":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/zai-org/GLM-4.7-Flash":{"mode":"chat","base_model":"GLM-4.7-Flash","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.7-Flash","max_input_tokens":200000,"max_tokens":200000,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":200000}}]},"huggingface/deepinfra/zai-org/GLM-4.7-Flash":{"mode":"chat","base_model":"GLM-4.7-Flash","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.7-Flash","max_input_tokens":202752,"max_tokens":202752,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"huggingface/novita/inclusionAI/Ling-3.0-flash":{"mode":"chat","base_model":"Ling-3.0-flash","provider":"huggingface","source":"https://huggingface.co/inclusionAI/Ling-3.0-flash","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/inclusionAI/Ling-3.0-flash":{"mode":"chat","base_model":"Ling-3.0-flash","provider":"huggingface","source":"https://huggingface.co/inclusionAI/Ling-3.0-flash","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/moonshotai/Kimi-K2.7-Code":{"mode":"chat","base_model":"Kimi-K2.7-Code","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/fireworks-ai/moonshotai/Kimi-K2.7-Code":{"mode":"chat","base_model":"Kimi-K2.7-Code","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/baseten/moonshotai/Kimi-K2.7-Code":{"mode":"chat","base_model":"Kimi-K2.7-Code","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","max_input_tokens":262000,"max_tokens":262000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262000}}]},"huggingface/deepinfra/moonshotai/Kimi-K2.7-Code":{"mode":"chat","base_model":"Kimi-K2.7-Code","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.7-Code","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/microsoft/phi-4":{"mode":"chat","base_model":"phi-4","provider":"huggingface","source":"https://huggingface.co/microsoft/phi-4","max_input_tokens":16384,"max_tokens":16384,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"huggingface/novita/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"Llama-4-Scout-17B-16E-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/nscale/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"Llama-4-Scout-17B-16E-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E-Instruct","max_input_tokens":890000,"max_tokens":890000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":890000}}]},"huggingface/deepinfra/meta-llama/Llama-4-Scout-17B-16E-Instruct":{"mode":"chat","base_model":"Llama-4-Scout-17B-16E-Instruct","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E-Instruct","max_input_tokens":327680,"max_tokens":327680,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":327680}}]},"huggingface/nscale/Qwen/Qwen2.5-Coder-32B-Instruct":{"mode":"chat","base_model":"Qwen2.5-Coder-32B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen2.5-Coder-32B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/Qwen/Qwen3.5-27B":{"mode":"chat","base_model":"Qwen3.5-27B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-27B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/Qwen/Qwen3.5-27B":{"mode":"chat","base_model":"Qwen3.5-27B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-27B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/XiaomiMiMo/MiMo-V2.5":{"mode":"chat","base_model":"MiMo-V2.5","provider":"huggingface","source":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","max_input_tokens":1048576,"max_tokens":1048576,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/deepinfra/XiaomiMiMo/MiMo-V2.5":{"mode":"chat","base_model":"MiMo-V2.5","provider":"huggingface","source":"https://huggingface.co/XiaomiMiMo/MiMo-V2.5","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/Qwen/Qwen3-30B-A3B":{"mode":"chat","base_model":"Qwen3-30B-A3B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-30B-A3B","max_input_tokens":40960,"max_tokens":40960,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"huggingface/novita/moonshotai/Kimi-K2.5":{"mode":"chat","base_model":"Kimi-K2.5","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.5","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/moonshotai/Kimi-K2.5":{"mode":"chat","base_model":"Kimi-K2.5","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.5","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/fireworks-ai/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4":{"mode":"chat","base_model":"NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","provider":"huggingface","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/Qwen/Qwen3.5-35B-A3B":{"mode":"chat","base_model":"Qwen3.5-35B-A3B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-35B-A3B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/Qwen/Qwen3.5-35B-A3B":{"mode":"chat","base_model":"Qwen3.5-35B-A3B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-35B-A3B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/Sao10K/L3-8B-Stheno-v3.2":{"mode":"chat","base_model":"L3-8B-Stheno-v3.2","provider":"huggingface","source":"https://huggingface.co/Sao10K/L3-8B-Stheno-v3.2","max_input_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"huggingface/deepinfra/google/gemma-3-12b-it":{"mode":"chat","base_model":"gemma-3-12b-it","provider":"huggingface","source":"https://huggingface.co/google/gemma-3-12b-it","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/publicai/speakleash/Bielik-11B-v3.0-Instruct":{"mode":"chat","base_model":"Bielik-11B-v3.0-Instruct","provider":"huggingface","source":"https://huggingface.co/speakleash/Bielik-11B-v3.0-Instruct","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/nscale/Qwen/Qwen3-14B":{"mode":"chat","base_model":"Qwen3-14B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-14B","max_input_tokens":40960,"max_tokens":40960,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"huggingface/deepinfra/Qwen/Qwen3-14B":{"mode":"chat","base_model":"Qwen3-14B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-14B","max_input_tokens":40960,"max_tokens":40960,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"huggingface/nscale/Qwen/Qwen2.5-Coder-7B-Instruct":{"mode":"chat","base_model":"Qwen2.5-Coder-7B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen2.5-Coder-7B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/moonshotai/Kimi-K2-Instruct":{"mode":"chat","base_model":"Kimi-K2-Instruct","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/google/gemma-3-27b-it":{"mode":"chat","base_model":"gemma-3-27b-it","provider":"huggingface","source":"https://huggingface.co/google/gemma-3-27b-it","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/deepseek-ai/DeepSeek-V3.2":{"mode":"chat","base_model":"DeepSeek-V3.2","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.2","max_input_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V3.2":{"mode":"chat","base_model":"DeepSeek-V3.2","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.2","max_input_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-7B":{"mode":"chat","base_model":"DeepSeek-R1-Distill-Qwen-7B","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-7B","max_input_tokens":131072,"max_tokens":131072,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/Qwen/Qwen3-VL-235B-A22B-Instruct":{"mode":"chat","base_model":"Qwen3-VL-235B-A22B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/Qwen/Qwen3-VL-235B-A22B-Instruct":{"mode":"chat","base_model":"Qwen3-VL-235B-A22B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Instruct","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16":{"mode":"chat","base_model":"NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","provider":"huggingface","source":"https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"Kimi-K2.6","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.6","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/fireworks-ai/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"Kimi-K2.6","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.6","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/baseten/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"Kimi-K2.6","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.6","max_input_tokens":262000,"max_tokens":262000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262000}}]},"huggingface/deepinfra/moonshotai/Kimi-K2.6":{"mode":"chat","base_model":"Kimi-K2.6","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2.6","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/Qwen/Qwen3-VL-30B-A3B-Instruct":{"mode":"chat","base_model":"Qwen3-VL-30B-A3B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-VL-30B-A3B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/Qwen/Qwen3-VL-30B-A3B-Instruct":{"mode":"chat","base_model":"Qwen3-VL-30B-A3B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-VL-30B-A3B-Instruct","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/tencent/Hy3":{"mode":"chat","base_model":"Hy3","provider":"huggingface","source":"https://huggingface.co/tencent/Hy3","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/stepfun-ai/Step-3.7-Flash":{"mode":"chat","base_model":"Step-3.7-Flash","provider":"huggingface","source":"https://huggingface.co/stepfun-ai/Step-3.7-Flash","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/ibm-granite/granite-4.2-3b":{"mode":"chat","base_model":"granite-4.2-3b","provider":"huggingface","source":"https://huggingface.co/ibm-granite/granite-4.2-3b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"mode":"chat","base_model":"Llama-4-Maverick-17B-128E-Instruct-FP8","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","max_input_tokens":1048576,"max_tokens":1048576,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"huggingface/novita/deepseek-ai/DeepSeek-R1-Distill-Llama-70B":{"mode":"chat","base_model":"DeepSeek-R1-Distill-Llama-70B","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-70B","max_input_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"huggingface/deepinfra/meta-llama/Llama-Guard-4-12B":{"mode":"chat","base_model":"Llama-Guard-4-12B","provider":"huggingface","source":"https://huggingface.co/meta-llama/Llama-Guard-4-12B","max_input_tokens":163840,"max_tokens":163840,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/nscale/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B":{"mode":"chat","base_model":"DeepSeek-R1-Distill-Qwen-14B","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-14B","max_input_tokens":131072,"max_tokens":131072,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/Qwen/Qwen3.5-122B-A10B":{"mode":"chat","base_model":"Qwen3.5-122B-A10B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/Qwen/Qwen3.5-122B-A10B":{"mode":"chat","base_model":"Qwen3.5-122B-A10B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-122B-A10B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/Qwen/Qwen3-Coder-480B-A35B-Instruct":{"mode":"chat","base_model":"Qwen3-Coder-480B-A35B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/zai-org/GLM-5":{"mode":"chat","base_model":"GLM-5","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5","max_input_tokens":202800,"max_tokens":202800,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202800}}]},"huggingface/deepinfra/zai-org/GLM-5":{"mode":"chat","base_model":"GLM-5","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5","max_input_tokens":202752,"max_tokens":202752,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"huggingface/novita/MiniMaxAI/MiniMax-M2.5":{"mode":"chat","base_model":"MiniMax-M2.5","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.5","max_input_tokens":204800,"max_tokens":204800,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V3":{"mode":"chat","base_model":"DeepSeek-V3","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3","max_input_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V3-0324":{"mode":"chat","base_model":"DeepSeek-V3-0324","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3-0324","max_input_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/deepinfra/ibm-granite/granite-4.2-8b":{"mode":"chat","base_model":"granite-4.2-8b","provider":"huggingface","source":"https://huggingface.co/ibm-granite/granite-4.2-8b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/ibm-granite/granite-4.2-30b":{"mode":"chat","base_model":"granite-4.2-30b","provider":"huggingface","source":"https://huggingface.co/ibm-granite/granite-4.2-30b","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/nscale/Qwen/Qwen2.5-Coder-3B-Instruct":{"mode":"chat","base_model":"Qwen2.5-Coder-3B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen2.5-Coder-3B-Instruct","max_input_tokens":32768,"max_tokens":32768,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"huggingface/novita/Qwen/Qwen3.5-397B-A17B":{"mode":"chat","base_model":"Qwen3.5-397B-A17B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/scaleway/Qwen/Qwen3.5-397B-A17B":{"mode":"chat","base_model":"Qwen3.5-397B-A17B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/ovhcloud/Qwen/Qwen3.5-397B-A17B":{"mode":"chat","base_model":"Qwen3.5-397B-A17B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/deepinfra/Qwen/Qwen3.5-397B-A17B":{"mode":"chat","base_model":"Qwen3.5-397B-A17B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3.5-397B-A17B","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/Sao10K/L3-8B-Lunaris-v1":{"mode":"chat","base_model":"L3-8B-Lunaris-v1","provider":"huggingface","source":"https://huggingface.co/Sao10K/L3-8B-Lunaris-v1","max_input_tokens":8192,"max_tokens":8192,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"huggingface/novita/Qwen/Qwen2.5-72B-Instruct":{"mode":"chat","base_model":"Qwen2.5-72B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen2.5-72B-Instruct","max_input_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"huggingface/deepinfra/Qwen/Qwen2.5-72B-Instruct":{"mode":"chat","base_model":"Qwen2.5-72B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen2.5-72B-Instruct","max_input_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"huggingface/nscale/Qwen/Qwen3-32B":{"mode":"chat","base_model":"Qwen3-32B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-32B","max_input_tokens":40960,"max_tokens":40960,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"huggingface/deepinfra/Qwen/Qwen3-32B":{"mode":"chat","base_model":"Qwen3-32B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-32B","max_input_tokens":40960,"max_tokens":40960,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"huggingface/novita/moonshotai/Kimi-K2-Instruct-0905":{"mode":"chat","base_model":"Kimi-K2-Instruct-0905","provider":"huggingface","source":"https://huggingface.co/moonshotai/Kimi-K2-Instruct-0905","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/Qwen/Qwen3-Next-80B-A3B-Instruct":{"mode":"chat","base_model":"Qwen3-Next-80B-A3B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Instruct","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/Qwen/Qwen3-Next-80B-A3B-Instruct":{"mode":"chat","base_model":"Qwen3-Next-80B-A3B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Instruct","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/nscale/Qwen/Qwen3-4B-Thinking-2507":{"mode":"chat","base_model":"Qwen3-4B-Thinking-2507","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-4B-Thinking-2507","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/MiniMaxAI/MiniMax-M2.7":{"mode":"chat","base_model":"MiniMax-M2.7","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7","max_input_tokens":204800,"max_tokens":204800,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"huggingface/deepinfra/MiniMaxAI/MiniMax-M2.7":{"mode":"chat","base_model":"MiniMax-M2.7","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7","max_input_tokens":196608,"max_tokens":196608,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":196608}}]},"huggingface/deepinfra/stepfun-ai/Step-3.5-Flash":{"mode":"chat","base_model":"Step-3.5-Flash","provider":"huggingface","source":"https://huggingface.co/stepfun-ai/Step-3.5-Flash","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/Qwen/Qwen3-235B-A22B-Instruct-2507":{"mode":"chat","base_model":"Qwen3-235B-A22B-Instruct-2507","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/nscale/Qwen/Qwen3-235B-A22B-Instruct-2507":{"mode":"chat","base_model":"Qwen3-235B-A22B-Instruct-2507","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507","max_input_tokens":32768,"max_tokens":32768,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"huggingface/scaleway/Qwen/Qwen3-235B-A22B-Instruct-2507":{"mode":"chat","base_model":"Qwen3-235B-A22B-Instruct-2507","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/deepinfra/Qwen/Qwen3-235B-A22B-Instruct-2507":{"mode":"chat","base_model":"Qwen3-235B-A22B-Instruct-2507","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/novita/Qwen/Qwen3-VL-235B-A22B-Thinking":{"mode":"chat","base_model":"Qwen3-VL-235B-A22B-Thinking","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Thinking","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/zai-org/GLM-4.5-Air":{"mode":"chat","base_model":"GLM-4.5-Air","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.5-Air","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/zai-org/AutoGLM-Phone-9B-Multilingual":{"mode":"chat","base_model":"AutoGLM-Phone-9B-Multilingual","provider":"huggingface","source":"https://huggingface.co/zai-org/AutoGLM-Phone-9B-Multilingual","max_input_tokens":65536,"max_tokens":65536,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"huggingface/novita/baidu/ERNIE-4.5-VL-424B-A47B-Base-PT":{"mode":"chat","base_model":"ERNIE-4.5-VL-424B-A47B-Base-PT","provider":"huggingface","source":"https://huggingface.co/baidu/ERNIE-4.5-VL-424B-A47B-Base-PT","max_input_tokens":123000,"max_tokens":123000,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":123000}}]},"huggingface/novita/zai-org/GLM-4.5V":{"mode":"chat","base_model":"GLM-4.5V","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.5V","max_input_tokens":65536,"max_tokens":65536,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"huggingface/novita/zai-org/GLM-4.6V-Flash":{"mode":"chat","base_model":"GLM-4.6V-Flash","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.6V-Flash","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/zai-org/GLM-4.7":{"mode":"chat","base_model":"GLM-4.7","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.7","max_input_tokens":204800,"max_tokens":204800,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"huggingface/baseten/zai-org/GLM-4.7":{"mode":"chat","base_model":"GLM-4.7","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.7","max_input_tokens":200000,"max_tokens":200000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":200000}}]},"huggingface/deepinfra/zai-org/GLM-4.7":{"mode":"chat","base_model":"GLM-4.7","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.7","max_input_tokens":202752,"max_tokens":202752,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"huggingface/publicai/aisingapore/Gemma-SEA-LION-v4-27B-IT":{"mode":"chat","base_model":"Gemma-SEA-LION-v4-27B-IT","provider":"huggingface","source":"https://huggingface.co/aisingapore/Gemma-SEA-LION-v4-27B-IT","supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/deepinfra/NousResearch/Hermes-3-Llama-3.1-70B":{"mode":"chat","base_model":"Hermes-3-Llama-3.1-70B","provider":"huggingface","source":"https://huggingface.co/NousResearch/Hermes-3-Llama-3.1-70B","max_input_tokens":131072,"max_tokens":131072,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/publicai/swiss-ai/Apertus-8B-Instruct-2509":{"mode":"chat","base_model":"Apertus-8B-Instruct-2509","provider":"huggingface","source":"https://huggingface.co/swiss-ai/Apertus-8B-Instruct-2509","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/novita/deepseek-ai/DeepSeek-R1-0528":{"mode":"chat","base_model":"DeepSeek-R1-0528","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-R1-0528","max_input_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-R1-0528":{"mode":"chat","base_model":"DeepSeek-R1-0528","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-R1-0528","max_input_tokens":163840,"max_tokens":163840,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/publicai/swiss-ai/Apertus-70B-Instruct-2509":{"mode":"chat","base_model":"Apertus-70B-Instruct-2509","provider":"huggingface","source":"https://huggingface.co/swiss-ai/Apertus-70B-Instruct-2509","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/novita/Qwen/Qwen3-235B-A22B-Thinking-2507":{"mode":"chat","base_model":"Qwen3-235B-A22B-Thinking-2507","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Thinking-2507","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/Qwen/Qwen3-235B-A22B-Thinking-2507":{"mode":"chat","base_model":"Qwen3-235B-A22B-Thinking-2507","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-235B-A22B-Thinking-2507","max_input_tokens":262144,"max_tokens":262144,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"huggingface/publicai/aisingapore/Qwen-SEA-LION-v4-32B-IT":{"mode":"chat","base_model":"Qwen-SEA-LION-v4-32B-IT","provider":"huggingface","source":"https://huggingface.co/aisingapore/Qwen-SEA-LION-v4-32B-IT","supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"huggingface/novita/deepseek-ai/DeepSeek-V3.1":{"mode":"chat","base_model":"DeepSeek-V3.1","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.1","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V3.1":{"mode":"chat","base_model":"DeepSeek-V3.1","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.1","max_input_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/ovhcloud/Qwen/Qwen2.5-VL-72B-Instruct":{"mode":"chat","base_model":"Qwen2.5-VL-72B-Instruct","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen2.5-VL-72B-Instruct","max_input_tokens":32768,"max_tokens":32768,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"huggingface/novita/deepseek-ai/DeepSeek-V3.1-Terminus":{"mode":"chat","base_model":"DeepSeek-V3.1-Terminus","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.1-Terminus","max_input_tokens":131072,"max_tokens":131072,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/deepinfra/deepseek-ai/DeepSeek-V3.1-Terminus":{"mode":"chat","base_model":"DeepSeek-V3.1-Terminus","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.1-Terminus","max_input_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/nscale/deepseek-ai/DeepSeek-R1-Distill-Llama-8B":{"mode":"chat","base_model":"DeepSeek-R1-Distill-Llama-8B","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-8B","max_input_tokens":131072,"max_tokens":131072,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"huggingface/novita/alpindale/WizardLM-2-8x22B":{"mode":"chat","base_model":"WizardLM-2-8x22B","provider":"huggingface","source":"https://huggingface.co/alpindale/WizardLM-2-8x22B","max_input_tokens":65535,"max_tokens":65535,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65535}}]},"huggingface/novita/zai-org/GLM-4.6":{"mode":"chat","base_model":"GLM-4.6","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.6","max_input_tokens":204800,"max_tokens":204800,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"huggingface/deepinfra/zai-org/GLM-4.6":{"mode":"chat","base_model":"GLM-4.6","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4.6","max_input_tokens":202752,"max_tokens":202752,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"huggingface/novita/Qwen/Qwen3-235B-A22B":{"mode":"chat","base_model":"Qwen3-235B-A22B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-235B-A22B","max_input_tokens":40960,"max_tokens":40960,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40960}}]},"huggingface/nscale/Qwen/Qwen3-235B-A22B":{"mode":"chat","base_model":"Qwen3-235B-A22B","provider":"huggingface","source":"https://huggingface.co/Qwen/Qwen3-235B-A22B","max_input_tokens":32000,"max_tokens":32000,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"huggingface/novita/zai-org/GLM-4-32B-0414":{"mode":"chat","base_model":"GLM-4-32B-0414","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-4-32B-0414","max_input_tokens":32000,"max_tokens":32000,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"huggingface/deepinfra/zai-org/GLM-5.1":{"mode":"chat","base_model":"GLM-5.1","provider":"huggingface","source":"https://huggingface.co/zai-org/GLM-5.1","max_input_tokens":202752,"max_tokens":202752,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"huggingface/novita/MiniMaxAI/MiniMax-M1-80k":{"mode":"chat","base_model":"MiniMax-M1-80k","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M1-80k","max_input_tokens":1000000,"max_tokens":1000000,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"huggingface/novita/deepseek-ai/DeepSeek-V3.2-Exp":{"mode":"chat","base_model":"DeepSeek-V3.2-Exp","provider":"huggingface","source":"https://huggingface.co/deepseek-ai/DeepSeek-V3.2-Exp","max_input_tokens":163840,"max_tokens":163840,"supports_function_calling":true,"supports_tool_choice":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":163840}}]},"huggingface/novita/MiniMaxAI/MiniMax-M2":{"mode":"chat","base_model":"MiniMax-M2","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M2","max_input_tokens":204800,"max_tokens":204800,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"huggingface/novita/MiniMaxAI/MiniMax-M2.1":{"mode":"chat","base_model":"MiniMax-M2.1","provider":"huggingface","source":"https://huggingface.co/MiniMaxAI/MiniMax-M2.1","max_input_tokens":204800,"max_tokens":204800,"supports_function_calling":true,"supports_tool_choice":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":204800}}]},"wafer/GLM-5.2":{"mode":"chat","base_model":"GLM-5.2","provider":"wafer","source":"https://pass.wafer.ai/v1/models","supports_prompt_caching":true,"max_input_tokens":1048576,"max_tokens":1048576,"supports_vision":false,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wafer/GLM-5.3":{"mode":"chat","base_model":"glm-5.3","provider":"wafer","source":"https://pass.wafer.ai/v1/models","supports_prompt_caching":true,"max_input_tokens":1048576,"max_tokens":1048576,"supports_vision":false,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wafer/DeepSeek-V4-Pro":{"mode":"chat","base_model":"deepseek-v4-pro","provider":"wafer","source":"https://pass.wafer.ai/v1/models","supports_prompt_caching":true,"max_input_tokens":1048576,"max_tokens":1048576,"supports_vision":false,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wafer/Kimi-K3":{"mode":"chat","base_model":"Kimi-K3","provider":"wafer","source":"https://pass.wafer.ai/v1/models","supports_prompt_caching":true,"max_input_tokens":1048576,"max_tokens":1048576,"max_output_tokens":131072,"supports_vision":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"wafer/Qwen3.8-27B":{"mode":"chat","base_model":"qwen3.8-27b","provider":"wafer","source":"https://pass.wafer.ai/v1/models","supports_prompt_caching":true,"max_input_tokens":262144,"max_tokens":262144,"supports_vision":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wafer/GLM-5.3-Flash":{"mode":"chat","base_model":"glm-5.3-flash","provider":"wafer","source":"https://pass.wafer.ai/v1/models","supports_prompt_caching":true,"max_input_tokens":1048576,"max_tokens":1048576,"supports_vision":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wafer/DeepSeek-V4.1-Flash":{"mode":"chat","base_model":"deepseek-v4.1-flash","provider":"wafer","source":"https://pass.wafer.ai/v1/models","supports_prompt_caching":true,"max_input_tokens":1048576,"max_tokens":1048576,"supports_vision":true,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wafer/DeepSeek-V4-Flash-0731-Fast":{"mode":"chat","base_model":"DeepSeek-V4-Flash-0731-Fast","provider":"wafer","source":"https://pass.wafer.ai/v1/models","supports_prompt_caching":true,"max_input_tokens":1048576,"max_tokens":1048576,"supports_vision":false,"supports_function_calling":true,"supports_reasoning":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"bedrock/mistral.pixtral-large-2502-v1:0":{"mode":"chat","base_model":"pixtral-large","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_reasoning":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock_mantle/mistral.pixtral-large-2502-v1:0":{"mode":"chat","base_model":"pixtral-large","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":8192,"max_tokens":8192,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_reasoning":false,"provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"bedrock/us.openai.gpt-6-luna":{"mode":"chat","base_model":"gpt-6-luna","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"knowledge_cutoff":"2026-05-18","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us.openai.gpt-6-luna":{"mode":"chat","base_model":"gpt-6-luna","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"knowledge_cutoff":"2026-05-18","provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/deepseek.r1-v1:0":{"mode":"chat","base_model":"deepseek-r1","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":32768,"max_tokens":32768,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_reasoning":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"bedrock/jp.amazon.nova-2-lite-v1:0":{"mode":"chat","base_model":"nova-2-lite","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1000000,"max_output_tokens":65535,"max_tokens":65535,"supported_modalities":["text","image","video","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_video_input":true,"supports_function_calling":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65535}}]},"bedrock_mantle/jp.amazon.nova-2-lite-v1:0":{"mode":"chat","base_model":"nova-2-lite","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1000000,"max_output_tokens":65535,"max_tokens":65535,"supported_modalities":["text","image","video","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_video_input":true,"supports_function_calling":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65535}}]},"bedrock/in.openai.gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"knowledge_cutoff":"2026-02-16","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/in.openai.gpt-5.6-terra":{"mode":"chat","base_model":"gpt-5.6-terra","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"knowledge_cutoff":"2026-02-16","provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/us.openai.gpt-6-sol":{"mode":"chat","base_model":"gpt-6-sol","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"knowledge_cutoff":"2026-04-20","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us.openai.gpt-6-sol":{"mode":"chat","base_model":"gpt-6-sol","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"knowledge_cutoff":"2026-04-20","provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock/ca.amazon.nova-lite-v1:0":{"mode":"chat","base_model":"nova-lite","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supported_modalities":["text","image","video","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_video_input":true,"supports_function_calling":true,"supports_reasoning":false,"provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":10000}}]},"bedrock_mantle/ca.amazon.nova-lite-v1:0":{"mode":"chat","base_model":"nova-lite","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":300000,"max_output_tokens":10000,"max_tokens":10000,"supported_modalities":["text","image","video","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_video_input":true,"supports_function_calling":true,"supports_reasoning":false,"provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":10000}}]},"bedrock/in.openai.gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"knowledge_cutoff":"2026-02-16","provider":"bedrock","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/in.openai.gpt-5.6-luna":{"mode":"chat","base_model":"gpt-5.6-luna","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":922000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh","max"],"knowledge_cutoff":"2026-02-16","provider":"bedrock_mantle","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"vertex_ai/gemini-flash-lite-latest":{"mode":"chat","base_model":"gemini-flash-lite","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"supported_modalities":["text","image","video","audio","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_video_input":true,"supports_function_calling":true,"supports_reasoning":true,"reasoning_effort_levels":["minimal","low","medium","high"],"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"vertex_ai/gemini-flash-latest":{"mode":"chat","base_model":"gemini-flash","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1048576,"max_output_tokens":65536,"max_tokens":65536,"supported_modalities":["text","image","video","audio","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_audio_input":true,"supports_video_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["minimal","low","medium","high"],"provider":"vertex_ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":65536}}]},"gpt-5.3-codex-spark":{"mode":"chat","base_model":"gpt-5.3-codex-spark","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":100000,"max_output_tokens":32000,"max_tokens":32000,"supported_modalities":["text","image","pdf"],"supported_output_modalities":["text"],"supports_vision":true,"supports_pdf_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","medium","high","xhigh"],"knowledge_cutoff":"2025-08-31","provider":"openai","model_parameters":[{"id":"max_completion_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"cloudflare/workers-ai/@cf/qwen/qwen3-30b-a3b-fp8":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"workers-ai/workers-ai/@cf/qwen/qwen3-30b-a3b-fp8":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":true,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"cloudflare/workers-ai/@cf/qwen/qwen3.8-27b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","xhigh"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"workers-ai/workers-ai/@cf/qwen/qwen3.8-27b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","xhigh"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"cloudflare/workers-ai/@cf/qwen/qwq-32b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":24000,"max_output_tokens":24000,"max_tokens":24000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":24000}}]},"workers-ai/workers-ai/@cf/qwen/qwq-32b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":24000,"max_output_tokens":24000,"max_tokens":24000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":true,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":24000}}]},"cloudflare/workers-ai/@cf/qwen/qwen2.5-coder-32b-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"workers-ai/workers-ai/@cf/qwen/qwen2.5-coder-32b-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32768}}]},"cloudflare/workers-ai/@cf/deepseek-ai/deepseek-v4-pro-0813":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","high","max"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"workers-ai/workers-ai/@cf/deepseek-ai/deepseek-v4-pro-0813":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1048576,"max_output_tokens":1048576,"max_tokens":1048576,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","high","max"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"cloudflare/workers-ai/@cf/deepseek-ai/deepseek-v4-flash-0731":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1310720,"max_output_tokens":1048576,"max_tokens":1048576,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","high","max"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"workers-ai/workers-ai/@cf/deepseek-ai/deepseek-v4-flash-0731":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1310720,"max_output_tokens":1048576,"max_tokens":1048576,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","low","high","max"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"cloudflare/workers-ai/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":80000,"max_output_tokens":80000,"max_tokens":80000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":80000}}]},"workers-ai/workers-ai/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":80000,"max_output_tokens":80000,"max_tokens":80000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":true,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":80000}}]},"cloudflare/workers-ai/@cf/mistralai/mistral-small-3.1-24b-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"workers-ai/workers-ai/@cf/mistralai/mistral-small-3.1-24b-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/workers-ai/@cf/nvidia/nemotron-3-120b-a12b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"workers-ai/workers-ai/@cf/nvidia/nemotron-3-120b-a12b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":256000,"max_output_tokens":256000,"max_tokens":256000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"cloudflare/workers-ai/@cf/google/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":256000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","high"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"workers-ai/workers-ai/@cf/google/gemma-4-26b-a4b-it":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":256000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","high"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"cloudflare/workers-ai/@cf/zai-org/glm-5.2":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":256000,"max_tokens":256000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","high","max"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"workers-ai/workers-ai/@cf/zai-org/glm-5.2":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":256000,"max_tokens":256000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","high","max"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"cloudflare/workers-ai/@cf/zai-org/glm-5.3-flash":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1310720,"max_output_tokens":1048576,"max_tokens":1048576,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","high","max"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"workers-ai/workers-ai/@cf/zai-org/glm-5.3-flash":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1310720,"max_output_tokens":1048576,"max_tokens":1048576,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","high","max"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"cloudflare/workers-ai/@cf/zai-org/glm-5.3":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1310720,"max_output_tokens":1048576,"max_tokens":1048576,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","high","max"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"workers-ai/workers-ai/@cf/zai-org/glm-5.3":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":1310720,"max_output_tokens":1048576,"max_tokens":1048576,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","high","max"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"cloudflare/workers-ai/@cf/zai-org/glm-4.7-flash":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"workers-ai/workers-ai/@cf/zai-org/glm-4.7-flash":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"cloudflare/workers-ai/@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"workers-ai/workers-ai/@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/workers-ai/@cf/meta/llama-guard-3-8b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"workers-ai/workers-ai/@cf/meta/llama-guard-3-8b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":131072,"max_output_tokens":131072,"max_tokens":131072,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131072}}]},"cloudflare/workers-ai/@cf/meta/llama-3.1-8b-instruct-fp8":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"workers-ai/workers-ai/@cf/meta/llama-3.1-8b-instruct-fp8":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"cloudflare/workers-ai/@cf/meta/llama-3.2-3b-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":80000,"max_output_tokens":80000,"max_tokens":80000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":80000}}]},"workers-ai/workers-ai/@cf/meta/llama-3.2-3b-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":80000,"max_output_tokens":80000,"max_tokens":80000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":80000}}]},"cloudflare/workers-ai/@cf/meta/llama-3.2-1b-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":60000,"max_output_tokens":60000,"max_tokens":60000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":60000}}]},"workers-ai/workers-ai/@cf/meta/llama-3.2-1b-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":60000,"max_output_tokens":60000,"max_tokens":60000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":60000}}]},"cloudflare/workers-ai/@cf/meta/llama-4-scout-17b-16e-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":131000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"workers-ai/workers-ai/@cf/meta/llama-4-scout-17b-16e-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":131000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"cloudflare/workers-ai/@cf/meta/llama-3.3-70b-instruct-fp8-fast":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":24000,"max_output_tokens":24000,"max_tokens":24000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":24000}}]},"workers-ai/workers-ai/@cf/meta/llama-3.3-70b-instruct-fp8-fast":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":24000,"max_output_tokens":24000,"max_tokens":24000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":24000}}]},"cloudflare/workers-ai/@cf/meta/llama-3.2-11b-vision-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"workers-ai/workers-ai/@cf/meta/llama-3.2-11b-vision-instruct":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":128000,"max_tokens":128000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":false,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"cloudflare/workers-ai/@cf/ibm-granite/granite-4.0-h-micro":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":false,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131000}}]},"workers-ai/workers-ai/@cf/ibm-granite/granite-4.0-h-micro":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":131000,"max_output_tokens":131000,"max_tokens":131000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":false,"supports_reasoning":false,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":131000}}]},"cloudflare/workers-ai/@cf/openai/gpt-oss-20b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"workers-ai/workers-ai/@cf/openai/gpt-oss-20b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"cloudflare/workers-ai/@cf/openai/gpt-oss-120b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"workers-ai/workers-ai/@cf/openai/gpt-oss-120b":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","max_input_tokens":128000,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16384}}]},"cloudflare/workers-ai/@cf/moonshotai/kimi-k2.6":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":256000,"max_tokens":256000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","high"],"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"workers-ai/workers-ai/@cf/moonshotai/kimi-k2.6":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":256000,"max_tokens":256000,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["none","high"],"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":256000}}]},"cloudflare/workers-ai/@cf/moonshotai/kimi-k2.7-code":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"cloudflare","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"workers-ai/workers-ai/@cf/moonshotai/kimi-k2.7-code":{"mode":"chat","base_model":"cloudflare/workers-ai/","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_vision":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"provider":"workers-ai","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"databricks/databricks-kimi-k2-7-code":{"mode":"chat","base_model":"databricks/databricks-kimi-k2-7-code","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_vision":true,"supports_video_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"databricks/system.ai.kimi-k2-7-code":{"mode":"chat","base_model":"databricks/databricks-kimi-k2-7-code","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_vision":true,"supports_video_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"databricks/kimi-k2-7-code":{"mode":"chat","base_model":"databricks/databricks-kimi-k2-7-code","source":"https://models.dev/api.json","supports_prompt_caching":true,"max_input_tokens":262144,"max_output_tokens":262144,"max_tokens":262144,"supported_modalities":["text","image","video"],"supported_output_modalities":["text"],"supports_vision":true,"supports_video_input":true,"supports_function_calling":true,"supports_response_schema":true,"supports_reasoning":true,"reasoning_effort_levels":["low","medium","high"],"provider":"databricks","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"bedrock_mantle/mistral.ministral-3-8b-instruct":{"mode":"chat","base_model":"ministral-3-8b-instruct","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1"],"max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/mistral.ministral-3-14b-instruct":{"mode":"chat","base_model":"ministral-3-14b-instruct","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/zai.glm5":{"mode":"chat","base_model":"glm5","provider":"bedrock_mantle","source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"bedrock_mantle/google.gemma-3-12b-it":{"mode":"chat","base_model":"gemma-3-12b-it","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/writer.palmyra-vision-7b":{"mode":"chat","base_model":"palmyra-vision-7b","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","ap-northeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","sa-east-1"],"max_input_tokens":4000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"bedrock_mantle/moonshotai.kimi-k3":{"mode":"chat","base_model":"kimi-k3","provider":"bedrock_mantle","supports_prompt_caching":true,"supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_response_schema":true,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions","/v1/responses"],"supported_regions":["us-east-1","us-east-2","us-west-1","us-west-2","ca-central-1","ca-west-1","eu-central-1","eu-central-2","eu-north-1","eu-south-1","eu-south-2","eu-west-1","eu-west-2","eu-west-3","ap-east-2","ap-northeast-1","ap-northeast-2","ap-northeast-3","ap-south-1","ap-south-2","ap-southeast-1","ap-southeast-2","ap-southeast-3","ap-southeast-4","ap-southeast-5","ap-southeast-6","ap-southeast-7","il-central-1","me-central-1","me-south-1","af-south-1","sa-east-1"],"max_input_tokens":1000000,"max_tokens":1000000,"source":"https://aws.amazon.com/bedrock/pricing/","min_output_tokens":16,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1000000}}]},"bedrock_mantle/mistral.ministral-3-3b-instruct":{"mode":"chat","base_model":"ministral-3-3b-instruct","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/mistral.mistral-large-3-675b-instruct":{"mode":"chat","base_model":"mistral-large-3-675b-instruct","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1"],"max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock_mantle/deepseek.v3.2":{"mode":"chat","base_model":"v3.2","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-north-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1"],"max_input_tokens":164000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/mistral.voxtral-small-24b-2507":{"mode":"chat","base_model":"voxtral-small-24b-2507","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":32000,"max_tokens":32000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock_mantle/qwen.qwen3-next-80b-a3b-instruct":{"mode":"chat","base_model":"qwen3-next-80b-a3b-instruct","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1"],"max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/mistral.devstral-2-123b":{"mode":"chat","base_model":"devstral-2-123b","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4"],"max_input_tokens":256000,"max_output_tokens":32000,"max_tokens":32000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock_mantle/minimax.minimax-m2.1":{"mode":"chat","base_model":"minimax-m2.1","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4"],"max_input_tokens":196000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/qwen.qwen3-coder-next":{"mode":"chat","base_model":"qwen3-coder-next","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","eu-west-2","ap-southeast-2"],"max_input_tokens":256000,"max_output_tokens":16000,"max_tokens":16000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16000}}]},"bedrock_mantle/deepseek.v3.1":{"mode":"chat","base_model":"v3.1","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-2","us-west-2","eu-north-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","us-east-1"],"max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/qwen.qwen3-235b-a22b-2507":{"mode":"chat","base_model":"qwen3-235b-a22b-2507","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","eu-west-1","sa-east-1","us-east-1"],"max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/qwen.qwen3-32b":{"mode":"chat","base_model":"qwen3-32b","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1"],"max_input_tokens":32000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/zai.glm-4.7":{"mode":"chat","base_model":"glm-4.7","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-north-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4"],"max_input_tokens":203000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"bedrock_mantle/qwen.qwen3-vl-235b-a22b-instruct":{"mode":"chat","base_model":"qwen3-vl-235b-a22b-instruct","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1"],"max_input_tokens":256000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/qwen.qwen3-coder-30b-a3b-instruct":{"mode":"chat","base_model":"qwen3-coder-30b-a3b-instruct","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1"],"max_input_tokens":256000,"max_output_tokens":16000,"max_tokens":16000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16000}}]},"bedrock_mantle/google.gemma-3-27b-it":{"mode":"chat","base_model":"gemma-3-27b-it","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/mistral.voxtral-mini-3b-2507":{"mode":"chat","base_model":"voxtral-mini-3b-2507","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":true,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":32000,"max_tokens":32000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":32000}}]},"bedrock_mantle/minimax.minimax-m2":{"mode":"chat","base_model":"minimax-m2","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":1000000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/google.gemma-3-4b-it":{"mode":"chat","base_model":"gemma-3-4b-it","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":128000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/mistral.magistral-small-2509":{"mode":"chat","base_model":"magistral-small-2509","provider":"bedrock_mantle","supports_vision":true,"supports_audio_input":false,"supports_audio_output":false,"supports_reasoning":true,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","sa-east-1","ap-southeast-3","ap-southeast-4","eu-central-1","eu-north-1"],"max_input_tokens":128000,"max_output_tokens":40000,"max_tokens":40000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":40000}}]},"bedrock_mantle/zai.glm-4.7-flash":{"mode":"chat","base_model":"glm-4.7-flash","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4"],"max_input_tokens":203000,"max_output_tokens":4000,"max_tokens":4000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4000,"range":{"min":1,"max":4000}}]},"bedrock_mantle/minimax.minimax-m2.5":{"mode":"chat","base_model":"minimax-m2.5","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-1","us-east-2","us-west-2","eu-central-1","eu-north-1","eu-south-1","eu-west-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","ap-southeast-4"],"max_input_tokens":196000,"max_output_tokens":8000,"max_tokens":8000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8000}}]},"bedrock_mantle/qwen.qwen3-coder-480b-a35b-instruct":{"mode":"chat","base_model":"qwen3-coder-480b-a35b-instruct","provider":"bedrock_mantle","supports_vision":false,"supports_audio_input":false,"supports_audio_output":false,"supports_function_calling":true,"supports_tool_choice":true,"supported_endpoints":["/v1/chat/completions"],"supported_regions":["us-east-2","us-west-2","eu-north-1","eu-west-2","ap-northeast-1","ap-south-1","ap-southeast-2","ap-southeast-3","sa-east-1","us-east-1"],"max_input_tokens":128000,"max_output_tokens":16000,"max_tokens":16000,"source":"https://aws.amazon.com/bedrock/pricing/","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":16000}}]},"wafer/GLM-5.1":{"mode":"chat","base_model":"GLM-5.1","provider":"wafer","max_tokens":202752,"supports_function_calling":true,"supports_reasoning":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":202752}}]},"wafer/Kimi-K2.6":{"mode":"chat","base_model":"Kimi-K2.6","provider":"wafer","max_tokens":262144,"supports_function_calling":true,"supports_vision":true,"supports_reasoning":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262144}}]},"wafer/MiniMax-M3":{"mode":"chat","base_model":"MiniMax-M3","provider":"wafer","max_tokens":1048576,"supports_function_calling":true,"supports_vision":true,"supports_reasoning":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wafer/Qwen3.5-397B-A17B":{"mode":"chat","base_model":"Qwen3.5-397B-A17B","provider":"wafer","max_tokens":262133,"supports_function_calling":true,"supports_reasoning":true,"supports_system_messages":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":262133}}]},"wafer/glm5.2-fast":{"mode":"chat","base_model":"glm5.2-fast","provider":"wafer","max_tokens":1048576,"supports_function_calling":true,"supports_reasoning":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"wafer/kimi-k3-fast":{"mode":"chat","base_model":"kimi-k3-fast","provider":"wafer","max_tokens":1048576,"supports_function_calling":true,"supports_vision":true,"supports_reasoning":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_system_messages":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":1048576}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-5.6-sol":{"mode":"responses","base_model":"gpt-5.6-sol","max_input_tokens":1000000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"bedrock_mantle/us-gov-west-1/openai.gpt-5.5":{"mode":"responses","base_model":"gpt-5.5","max_input_tokens":272000,"max_output_tokens":128000,"max_tokens":128000,"use_openai_responses_path":true,"supported_endpoints":["/v1/responses"],"supported_modalities":["text","image"],"supported_output_modalities":["text"],"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_response_schema":true,"supports_tool_choice":true,"supports_vision":true,"provider":"bedrock_mantle","model_parameters":[{"id":"max_output_tokens","label":"Max Output Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":128000}}]},"runware/zai:glm@5.3-flash":{"mode":"chat","base_model":"glm-5.3-flash","provider":"runware","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"runware/google:gemini@3.7-flash":{"mode":"chat","base_model":"gemini-3.7-flash","provider":"runware","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"runware/deepseek:v4@flash":{"mode":"chat","base_model":"deepseek-v4-flash","provider":"runware","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","range":{"min":1}}]},"databricks/claude-haiku-4-5":{"mode":"chat","base_model":"claude-haiku-4-5","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"metadata":{"notes":"Input/output cost per token is dbu cost * $0.070. Number provided for reference, '*_dbu_cost_per_token' used in actual calculation. In-geo endpoint is 10% higher."},"source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_assistant_prefill":true,"supports_function_calling":true,"supports_prompt_caching":true,"supports_reasoning":true,"supports_anthropic_thinking_payload":true,"supports_tool_choice":true,"prompt_cache_min_tokens":4096,"provider":"databricks","supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"databricks/claude-sonnet-4-5":{"mode":"chat","base_model":"claude-sonnet-4-5","provider":"databricks","source":"https://www.databricks.com/product/pricing/proprietary-foundation-model-serving","supports_prompt_caching":true,"max_input_tokens":200000,"max_tokens":64000,"max_output_tokens":64000,"supports_vision":true,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":64000}}]},"google/gemma-4-26b-a4b-it-maas":{"mode":"chat","base_model":"gemma-4","max_input_tokens":262144,"max_output_tokens":8192,"max_tokens":8192,"source":"https://cloud.google.com/vertex-ai/generative-ai/docs/learn/models#foundation_models","supported_regions":["global"],"supports_function_calling":true,"supports_tool_choice":true,"supports_vision":true,"provider":"vertex_ai","tpm":250000,"rpm":10,"model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":8192}}]},"gemini-3.1-flash-lite-image":{"mode":"image_generation","base_model":"gemini-3.1-flash-lite-image","max_input_tokens":65536,"max_output_tokens":66000,"max_tokens":66000,"rpm":null,"source":"https://ai.google.dev/gemini-api/docs/pricing#gemini-3.1-flash-lite-image","supported_endpoints":["/v1/chat/completions","/v1/completions","/v1/batch"],"supported_modalities":["text","image","video"],"supported_output_modalities":["text","image"],"supports_function_calling":true,"supports_prompt_caching":false,"supports_reasoning":false,"supports_response_schema":true,"supports_system_messages":true,"supports_vision":true,"supports_web_search":false,"tpm":null,"provider":"gemini","model_parameters":[{"id":"max_tokens","label":"Max Tokens","helpText":"The maximum number of tokens that can be generated in the Result.","type":"number","default":4096,"range":{"min":1,"max":66000}}]},"aiml/openai/gpt-image-2":{"metadata":{"notes":"OpenAI gpt-image-2 via AI/ML API - flagship multimodal image generation and editing model with reasoning and 2K output. output_cost_per_image is AI/ML's published medium-quality rate; like the other aiml image entries it is billed as a flat per-image price"},"mode":"image_generation","source":"https://docs.aimlapi.com/api-references/image-models/openai/gpt-image-2","supported_endpoints":["/v1/images/generations"],"supports_vision":true,"provider":"aiml","base_model":"gpt-image-2"},"us.amazon.nova-canvas-v1:0":{"deprecation_date":"2026-09-30","max_input_tokens":2600,"mode":"image_generation","supports_nova_canvas_image_edit":true,"provider":"bedrock","base_model":"nova-canvas"},"amazon.nova-2-sonic-v1:0":{"mode":"realtime","supports_audio_input":true,"supports_audio_output":true,"provider":"bedrock","base_model":"nova-2-sonic"},"azure/gpt-image-2":{"deprecation_date":"2027-10-21","mode":"image_generation","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"azure","base_model":"gpt-image-2"},"azure/whisper":{"deprecation_date":"2026-12-15","mode":"audio_transcription","source":"https://azure.microsoft.com/en-us/pricing/details/cognitive-services/openai-service/","provider":"azure","base_model":"whisper"},"azure/gpt-realtime-1.5-2026-02-23":{"deprecation_date":"2027-08-24","max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"mode":"realtime","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","base_model":"gpt-realtime-1.5"},"azure/gpt-realtime-2.1":{"deprecation_date":"2027-06-25","max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"mode":"realtime","source":"https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","base_model":"gpt-realtime-2.1"},"azure/gpt-realtime-2.1-mini":{"deprecation_date":"2027-06-25","max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"mode":"realtime","source":"https://learn.microsoft.com/en-us/azure/foundry/foundry-models/concepts/models-sold-directly-by-azure","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","base_model":"gpt-realtime-2.1-mini"},"azure/gpt-realtime-mini":{"max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"mode":"realtime","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"azure","base_model":"gpt-realtime-mini"},"azure/gpt-realtime-whisper":{"mode":"audio_transcription","source":"https://learn.microsoft.com/en-us/azure/foundry/openai/concepts/gpt-realtime-whisper","supported_endpoints":["/v1/realtime","/v1/realtime/transcription_sessions"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"azure","base_model":"gpt-realtime-whisper"},"azure/gpt-image-2.5-flare":{"deprecation_date":"2027-09-09","mode":"image_generation","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"azure","base_model":"gpt-image-2.5-flare"},"azure/gpt-image-2.5-sunburst":{"deprecation_date":"2027-09-09","mode":"image_generation","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"azure","base_model":"gpt-image-2.5-sunburst"},"azure/gpt-image-2-2026-04-21":{"deprecation_date":"2027-10-21","mode":"image_generation","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"azure","base_model":"gpt-image-2"},"azure/speech/azure-stt":{"audio_transcription_config":"azure_speech","mode":"audio_transcription","source":"https://azure.microsoft.com/en-us/pricing/details/cognitive-services/speech-services/","supported_endpoints":["/v1/audio/transcriptions"],"provider":"azure","base_model":"speech/azure-stt"},"azure/FLUX.2-flex":{"max_input_tokens":32000,"max_tokens":32000,"mode":"image_generation","source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/black-forest-labs/","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supported_modalities":["text","image"],"supported_output_modalities":["image"],"provider":"azure","base_model":"flux.2-flex"},"azure/MAI-Image-2.5":{"mode":"image_generation","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"deprecation_date":"2026-10-01","provider":"azure","base_model":"mai-image-2.5"},"azure/MAI-Image-2.5-Flash":{"mode":"image_generation","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"deprecation_date":"2026-10-01","provider":"azure","base_model":"mai-image-2.5-flash"},"azure/MAI-Image-2.5-Pro":{"deprecation_date":"2026-10-01","mode":"image_generation","source":"https://techcommunity.microsoft.com/blog/azure-ai-foundry-blog/introducing-mai-image-2-5-pro-and-mai-voice-2-flash-in-microsoft-foundry/4539446","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"provider":"azure","base_model":"mai-image-2.5-pro"},"azure/mistral-document-ai-2512":{"mode":"ocr","supported_endpoints":["/v1/ocr"],"source":"https://ai.azure.com/catalog/models/mistral-document-ai-2512","provider":"azure","base_model":"mistral-document-ai"},"azure/mistral-ocr-4-0":{"mode":"ocr","supported_endpoints":["/v1/ocr"],"source":"https://azure.microsoft.com/en-us/pricing/details/ai-foundry-models/mistral/","provider":"azure","base_model":"mistral-ocr-4-0"},"azure/Cohere-parse-v5":{"deprecation_date":"2026-12-15","mode":"ocr","source":"https://cohere.com/blog/parse","supported_endpoints":["/v1/ocr"],"provider":"azure","base_model":"parse-v5"},"bedrock/guardrails":{"mode":"guardrail","source":"https://aws.amazon.com/bedrock/pricing/","provider":"bedrock","base_model":"guardrails"},"black_forest_labs/flux-kontext-pro":{"mode":"image_edit","source":"https://bfl.ai/pricing","supported_endpoints":["/v1/images/edits","/v1/images/generations"],"provider":"black_forest_labs","base_model":"flux-kontext-pro"},"black_forest_labs/flux-kontext-max":{"mode":"image_edit","source":"https://bfl.ai/pricing","supported_endpoints":["/v1/images/edits","/v1/images/generations"],"provider":"black_forest_labs","base_model":"flux-kontext-max"},"black_forest_labs/flux-pro-1.0-fill":{"mode":"image_edit","source":"https://bfl.ai/pricing","supported_endpoints":["/v1/images/edits"],"provider":"black_forest_labs","base_model":"flux-pro-1.0-fill"},"black_forest_labs/flux-pro-1.0-expand":{"mode":"image_edit","source":"https://bfl.ai/pricing","supported_endpoints":["/v1/images/edits"],"provider":"black_forest_labs","base_model":"flux-pro-1.0-expand"},"black_forest_labs/flux-pro-1.1":{"mode":"image_generation","source":"https://bfl.ai/pricing","supported_endpoints":["/v1/images/generations"],"provider":"black_forest_labs","base_model":"flux-pro-1.1"},"black_forest_labs/flux-pro-1.1-ultra":{"mode":"image_generation","source":"https://bfl.ai/pricing","supported_endpoints":["/v1/images/generations"],"provider":"black_forest_labs","base_model":"flux-pro-1.1-ultra"},"black_forest_labs/flux-dev":{"mode":"image_generation","source":"https://bfl.ai/pricing","supported_endpoints":["/v1/images/generations"],"provider":"black_forest_labs","base_model":"flux-dev"},"black_forest_labs/flux-pro":{"mode":"image_generation","source":"https://bfl.ai/pricing","supported_endpoints":["/v1/images/generations"],"provider":"black_forest_labs","base_model":"flux-pro"},"cohere/parse-v5.0":{"mode":"ocr","source":"https://cohere.com/blog/parse","supported_endpoints":["/v1/ocr"],"provider":"cohere","base_model":"parse"},"dashscope/qwen-image-2.0":{"mode":"image_generation","source":"https://www.alibabacloud.com/help/en/model-studio/models","supported_endpoints":["/v1/images/generations"],"provider":"dashscope","base_model":"qwen-image-2.0"},"dashscope/qwen-image-2.0-pro":{"mode":"image_generation","source":"https://www.alibabacloud.com/help/en/model-studio/models","supported_endpoints":["/v1/images/generations"],"provider":"dashscope","base_model":"qwen-image-2.0-pro"},"dashscope/qwen-image-3.0":{"mode":"image_generation","source":"https://www.alibabacloud.com/help/en/model-studio/models","supported_endpoints":["/v1/images/generations"],"provider":"dashscope","base_model":"qwen-image-3.0"},"dashscope/qwen-image-3.0-pro":{"mode":"image_generation","source":"https://www.alibabacloud.com/help/en/model-studio/models","supported_endpoints":["/v1/images/generations"],"provider":"dashscope","base_model":"qwen-image-3.0-pro"},"qwencloud/qwen-image-2.0":{"mode":"image_generation","source":"https://www.qwencloud.com/models","supported_endpoints":["/v1/images/generations"],"provider":"qwencloud","base_model":"qwen-image-2.0"},"qwencloud/qwen-image-2.0-pro":{"mode":"image_generation","source":"https://www.qwencloud.com/models","supported_endpoints":["/v1/images/generations"],"provider":"qwencloud","base_model":"qwen-image-2.0-pro"},"qwencloud/qwen-image-3.0":{"mode":"image_generation","source":"https://www.qwencloud.com/models","supported_endpoints":["/v1/images/generations"],"provider":"qwencloud","base_model":"qwen-image-3.0"},"qwencloud/qwen-image-3.0-pro":{"mode":"image_generation","source":"https://www.qwencloud.com/models","supported_endpoints":["/v1/images/generations"],"provider":"qwencloud","base_model":"qwen-image-3.0-pro"},"qwen_ai_platform/qwen-image-2.0":{"mode":"image_generation","source":"https://www.alibabacloud.com/help/en/model-studio/models","supported_endpoints":["/v1/images/generations"],"provider":"qwen_ai_platform","base_model":"qwen-image-2.0"},"qwen_ai_platform/qwen-image-2.0-pro":{"mode":"image_generation","source":"https://www.alibabacloud.com/help/en/model-studio/models","supported_endpoints":["/v1/images/generations"],"provider":"qwen_ai_platform","base_model":"qwen-image-2.0-pro"},"qwen_ai_platform/qwen-image-3.0":{"mode":"image_generation","source":"https://www.alibabacloud.com/help/en/model-studio/models","supported_endpoints":["/v1/images/generations"],"provider":"qwen_ai_platform","base_model":"qwen-image-3.0"},"qwen_ai_platform/qwen-image-3.0-pro":{"mode":"image_generation","source":"https://www.alibabacloud.com/help/en/model-studio/models","supported_endpoints":["/v1/images/generations"],"provider":"qwen_ai_platform","base_model":"qwen-image-3.0-pro"},"deepgram/streaming/nova-3":{"metadata":{"calculation":"$0.0048/60 seconds = $0.00008000 per second","note":"Nova-3 monolingual streaming, pay as you go","original_pricing_per_minute":0.0048},"mode":"audio_transcription","source":"https://deepgram.com/pricing","supported_endpoints":["/v1/listen"],"provider":"deepgram","base_model":"streaming/nova-3"},"deepgram/streaming/nova-3-multilingual":{"metadata":{"calculation":"$0.0058/60 seconds = $0.00009667 per second","note":"Nova-3 multilingual (language=multi) streaming, pay as you go","original_pricing_per_minute":0.0058},"mode":"audio_transcription","source":"https://deepgram.com/pricing","supported_endpoints":["/v1/listen"],"provider":"deepgram","base_model":"streaming/nova-3-multilingual"},"deepgram/streaming/redact":{"metadata":{"calculation":"$0.0020/60 seconds = $0.00003333 per second","note":"Redaction add-on (redact query param), streaming, pay as you go","original_pricing_per_minute":0.002},"mode":"audio_transcription","source":"https://deepgram.com/pricing","supported_endpoints":["/v1/listen"],"provider":"deepgram","base_model":"streaming/redact"},"deepgram/streaming/keyterm":{"metadata":{"calculation":"$0.0013/60 seconds = $0.00002167 per second","note":"Keyterm Prompting add-on (keyterm query param), streaming, pay as you go","original_pricing_per_minute":0.0013},"mode":"audio_transcription","source":"https://deepgram.com/pricing","supported_endpoints":["/v1/listen"],"provider":"deepgram","base_model":"streaming/keyterm"},"deepgram/streaming/detect_entities":{"metadata":{"calculation":"$0.0017/60 seconds = $0.00002833 per second","note":"Entity Detection add-on (detect_entities query param), streaming, pay as you go","original_pricing_per_minute":0.0017},"mode":"audio_transcription","source":"https://deepgram.com/pricing","supported_endpoints":["/v1/listen"],"provider":"deepgram","base_model":"streaming/detect-entities"},"deepgram/streaming/diarize":{"metadata":{"calculation":"$0.0020/60 seconds = $0.00003333 per second","note":"Speaker Diarization add-on (diarize / diarize_model query params), streaming, pay as you go","original_pricing_per_minute":0.002},"mode":"audio_transcription","source":"https://deepgram.com/pricing","supported_endpoints":["/v1/listen"],"provider":"deepgram","base_model":"streaming/diarize"},"serper/search":{"mode":"search","metadata":{"notes":"Serper Google Search API. Pricing: $1.00/1k queries (Starter), $0.75/1k (Standard), $0.50/1k (Scale), $0.30/1k (Ultimate)."},"provider":"serper","base_model":"search"},"apiserpent/search":{"mode":"search","metadata":{"notes":"APISerpent quick search (/api/search/quick), multi-engine (Google, Bing, Yahoo, DuckDuckGo). Pricing: $0.60/1k searches."},"provider":"apiserpent","base_model":"search"},"apiserpent/deep_search":{"mode":"search","metadata":{"notes":"APISerpent deep search (/api/search), multi-engine (Google, Bing, Yahoo, DuckDuckGo). Pricing: $0.60/1k searches."},"provider":"apiserpent","base_model":"deep-search"},"agentcore/search":{"mode":"search","metadata":{"notes":"Web Search on Amazon Bedrock AgentCore, billed by AWS on the gateway"},"provider":"agentcore","base_model":"search"},"bing_grounding/search":{"mode":"search","metadata":{"notes":"Grounding with Bing Search (G1 SKU): $35 per 1,000 transactions. Tokens for the Foundry model deployment that runs the grounded search are billed separately on that deployment."},"provider":"bing_grounding","base_model":"search"},"tinyfish/search":{"mode":"search","metadata":{"notes":"TinyFish Search API"},"provider":"tinyfish","base_model":"search"},"nimble/search":{"mode":"search","metadata":{"notes":"Nimble Search API pay-as-you-go list price: $5 per 1,000 searches, up to 100 results per search. Volume plans price differently."},"provider":"nimble","base_model":"search"},"fal_ai/bytedance/seedance-2.5/text-to-video":{"mode":"video_generation","source":"https://fal.ai/models/bytedance/seedance-2.5/text-to-video","supported_endpoints":["/v1/videos"],"supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"fal_ai","base_model":"seedance-2.5/text-to-video"},"fal_ai/bytedance/seedance-2.5/image-to-video":{"mode":"video_generation","source":"https://fal.ai/models/bytedance/seedance-2.5/image-to-video","supported_endpoints":["/v1/videos"],"supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"fal_ai","base_model":"seedance-2.5/image-to-video"},"fal_ai/bytedance/seedance-2.5/reference-to-video":{"mode":"video_generation","source":"https://fal.ai/models/bytedance/seedance-2.5/reference-to-video","supported_endpoints":["/v1/videos"],"supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"fal_ai","base_model":"seedance-2.5/reference-to-video"},"fal_ai/minimax/h3/text-to-video":{"mode":"video_generation","source":"https://fal.ai/models/minimax/h3/text-to-video","supported_endpoints":["/v1/videos"],"supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"fal_ai","base_model":"h3/text-to-video"},"fal_ai/minimax/h3/reference-to-video":{"mode":"video_generation","source":"https://fal.ai/models/minimax/h3/reference-to-video","supported_endpoints":["/v1/videos"],"supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"fal_ai","base_model":"h3/reference-to-video"},"fal_ai/bytedance/seedance-2.0/text-to-video":{"mode":"video_generation","source":"https://fal.ai/models/bytedance/seedance-2.0/text-to-video","metadata":{"comment":"fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160"},"supported_endpoints":["/v1/videos"],"supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"fal_ai","base_model":"seedance-2.0/text-to-video"},"fal_ai/bytedance/seedance-2.0/image-to-video":{"mode":"video_generation","source":"https://fal.ai/models/bytedance/seedance-2.0/image-to-video","metadata":{"comment":"fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160"},"supported_endpoints":["/v1/videos"],"supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"fal_ai","base_model":"seedance-2.0/image-to-video"},"fal_ai/bytedance/seedance-2.0/reference-to-video":{"mode":"video_generation","source":"https://fal.ai/models/bytedance/seedance-2.0/reference-to-video","metadata":{"comment":"fal bills $0.014 per 1k tokens (480p/720p/1080p) and $0.008 per 1k tokens (4k) with tokens = h*w*seconds*24/1024; 480p and 4k rates derived from that formula at 854x480 and 3840x2160"},"supported_endpoints":["/v1/videos"],"supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"fal_ai","base_model":"seedance-2.0/reference-to-video"},"fal_ai/fal-ai/nano-banana":{"mode":"image_generation","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","base_model":"nano-banana"},"fal_ai/fal-ai/gemini-25-flash-image":{"mode":"image_generation","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","base_model":"gemini-25-flash-image"},"fal_ai/openai/gpt-image-2":{"metadata":{"notes":""},"mode":"image_generation","source":"https://fal.ai/models/openai/gpt-image-2","supported_endpoints":["/v1/images/generations"],"supports_vision":true,"provider":"fal_ai","base_model":"gpt-image-2"},"fal_ai/gpt-image-2":{"metadata":{"notes":""},"mode":"image_generation","source":"https://fal.ai/models/openai/gpt-image-2","supported_endpoints":["/v1/images/generations"],"supports_vision":true,"provider":"fal_ai","base_model":"gpt-image-2"},"fal_ai/openai/gpt-image-2/edit":{"metadata":{"notes":"Editing endpoint of gpt-image-2 on fal.ai, reached through the image generation path with fal's image_urls param since /v1/images/edits is not wired for fal_ai. Prices include one input image and live in keyed entries fal_ai/{quality}/{width}-x-{height}/openai/gpt-image-2/edit. This flat entry is the fallback for the default edit request (quality=high, image_size=auto, inferred from the input image, priced as 1024x768 high)"},"mode":"image_generation","source":"https://fal.ai/models/openai/gpt-image-2/edit","supported_endpoints":["/v1/images/generations"],"supports_vision":true,"provider":"fal_ai","base_model":"edit"},"fal_ai/openai/gpt-image-2.5/flare/text-to-image":{"metadata":{"notes":"OpenAI gpt-image-2.5 (flare) served through fal.ai. fal publishes deterministic per-image prices per size and quality, mirrored as keyed entries fal_ai/{quality}/{width}-x-{height}/openai/gpt-image-2.5/flare/text-to-image that the fal_ai cost calculator picks from the request params. This flat entry is the fallback for the default request (quality=high, image_size=landscape_4_3 at 1024x768). quality=auto is priced as high"},"mode":"image_generation","source":"https://fal.ai/models/openai/gpt-image-2.5/flare/text-to-image","supported_endpoints":["/v1/images/generations"],"supports_vision":true,"provider":"fal_ai","base_model":"gpt-image-2.5/flare/text-to-image"},"fal_ai/openai/gpt-image-2.5/flare/edit":{"metadata":{"notes":"Editing endpoint of gpt-image-2.5 (flare) on fal.ai, reachable through /v1/images/edits or the image generation path with fal's image_urls param. Prices include one input image and live in keyed entries fal_ai/{quality}/{width}-x-{height}/openai/gpt-image-2.5/flare/edit. This flat entry is the fallback for the default edit request (quality=high, image_size=auto, inferred from the input image, priced as 1024x768 high)"},"mode":"image_generation","source":"https://fal.ai/models/openai/gpt-image-2.5/flare/edit","supported_endpoints":["/v1/images/edits","/v1/images/generations"],"supports_vision":true,"provider":"fal_ai","base_model":"gpt-image-2.5/flare/edit"},"fal_ai/openai/gpt-image-2.5/sunburst/text-to-image":{"metadata":{"notes":"OpenAI gpt-image-2.5 (sunburst) served through fal.ai. fal publishes deterministic per-image prices per size and quality, mirrored as keyed entries fal_ai/{quality}/{width}-x-{height}/openai/gpt-image-2.5/sunburst/text-to-image that the fal_ai cost calculator picks from the request params. This flat entry is the fallback for the default request (quality=high, image_size=landscape_4_3 at 1024x768). quality=auto is priced as high"},"mode":"image_generation","source":"https://fal.ai/models/openai/gpt-image-2.5/sunburst/text-to-image","supported_endpoints":["/v1/images/generations"],"supports_vision":true,"provider":"fal_ai","base_model":"gpt-image-2.5/sunburst/text-to-image"},"fal_ai/openai/gpt-image-2.5/sunburst/edit":{"metadata":{"notes":"Editing endpoint of gpt-image-2.5 (sunburst) on fal.ai, reachable through /v1/images/edits or the image generation path with fal's image_urls param. Prices include one input image and live in keyed entries fal_ai/{quality}/{width}-x-{height}/openai/gpt-image-2.5/sunburst/edit. This flat entry is the fallback for the default edit request (quality=high, image_size=auto, inferred from the input image, priced as 1024x768 high)"},"mode":"image_generation","source":"https://fal.ai/models/openai/gpt-image-2.5/sunburst/edit","supported_endpoints":["/v1/images/edits","/v1/images/generations"],"supports_vision":true,"provider":"fal_ai","base_model":"gpt-image-2.5/sunburst/edit"},"fal_ai/fal-ai/flux/dev":{"metadata":{"notes":"fal bills FLUX.1 [dev] at $0.025 per megapixel, rounding each image up to the nearest megapixel. The per-pixel rate is used when Fal reports the output size, and the flat per-image price is the fallback when dimensions are unavailable"},"mode":"image_generation","source":"https://fal.ai/models/fal-ai/flux/dev","supported_endpoints":["/v1/images/generations"],"provider":"fal_ai","base_model":"flux/dev"},"fal_ai/fal-ai/trellis":{"mode":"image_generation","source":"https://fal.ai/models/fal-ai/trellis","metadata":{"comment":"image-to-3D, returns a GLB mesh; served through the /fal_ai pass-through route"},"provider":"fal_ai","base_model":"trellis"},"fal_ai/fal-ai/trellis-2":{"mode":"image_generation","source":"https://fal.ai/models/fal-ai/trellis-2","metadata":{"comment":"image-to-3D, returns a GLB mesh; priced by the request's resolution field (default 1024); served through the /fal_ai pass-through route"},"provider":"fal_ai","base_model":"trellis-2"},"fal_ai/fal-ai/flux-lora-depth":{"metadata":{"notes":"fal bills fal-ai/flux-lora-depth at $0.035 per megapixel, rounding each image up to the nearest megapixel. The per-pixel rate is used when Fal reports the output size, and the flat per-image price prices the default 1 MP output like the sibling flux entries"},"mode":"image_generation","source":"https://fal.ai/models/fal-ai/flux-lora-depth","supported_endpoints":["/v1/images/edits"],"provider":"fal_ai","base_model":"flux-lora-depth"},"gemini-live-2.5-flash-native-audio":{"deprecation_date":"2026-12-13","max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"mode":"realtime","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/vertex_ai/live","/v1/realtime"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":true,"supports_tool_choice":true,"supports_url_context":false,"supports_vision":true,"supports_web_search":true,"gemini_native_audio":true,"provider":"vertex_ai","base_model":"gemini-live-2.5-flash-native-audio"},"gemini/gemini-omni-flash-preview":{"max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"mode":"video_generation","rpm":2000,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1beta/interactions"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","video"],"supports_audio_input":true,"supports_reasoning":true,"supports_system_messages":false,"supports_video_input":true,"supports_vision":true,"tpm":800000,"deprecation_date":"2026-09-30","provider":"gemini","base_model":"gemini-omni-flash-preview","supports_function_calling":false,"supports_parallel_function_calling":false,"supports_pdf_input":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_tool_choice":false},"gemini/veo-3.1-lite-generate-preview":{"max_input_tokens":1024,"max_tokens":1024,"mode":"video_generation","source":"https://ai.google.dev/gemini-api/docs/video","supported_modalities":["text"],"supported_output_modalities":["video"],"provider":"gemini","base_model":"veo-3.1-lite-generate"},"gpt-image-2":{"mode":"image_generation","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","base_model":"gpt-image-2"},"gpt-image-2-2026-04-21":{"mode":"image_generation","supported_endpoints":["/v1/images/generations","/v1/images/edits"],"supports_vision":true,"supports_pdf_input":true,"provider":"openai","base_model":"gpt-image-2"},"gpt-realtime-1.5":{"max_input_tokens":32000,"max_output_tokens":4096,"max_tokens":4096,"mode":"realtime","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","base_model":"gpt-realtime-1.5"},"gpt-realtime-2":{"max_input_tokens":128000,"max_output_tokens":32000,"max_tokens":32000,"mode":"realtime","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","base_model":"gpt-realtime-2"},"gpt-realtime-2.1":{"max_input_tokens":128000,"max_output_tokens":32000,"max_tokens":32000,"mode":"realtime","regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","base_model":"gpt-realtime-2.1"},"gpt-realtime-2.1-mini":{"max_input_tokens":128000,"max_output_tokens":32000,"max_tokens":32000,"mode":"realtime","regional_processing_uplift_multiplier_eu":1.1,"regional_processing_uplift_multiplier_us":1.1,"source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_system_messages":true,"supports_tool_choice":true,"provider":"openai","base_model":"gpt-realtime-2.1-mini"},"groq/canopylabs/orpheus-v1-english":{"max_input_tokens":4000,"max_output_tokens":50000,"max_tokens":50000,"mode":"audio_speech","source":"https://console.groq.com/docs/model/canopylabs/orpheus-v1-english","provider":"groq","base_model":"canopylabs/orpheus-v1-english"},"groq/canopylabs/orpheus-arabic-saudi":{"max_input_tokens":4000,"max_output_tokens":50000,"max_tokens":50000,"mode":"audio_speech","source":"https://console.groq.com/docs/models","provider":"groq","base_model":"canopylabs/orpheus-arabic-saudi"},"meta/muse-voice-transcribe-1.0":{"mode":"audio_transcription","source":"https://dev.meta.ai/docs/speech-to-text","supported_endpoints":["/v1/realtime"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"meta","base_model":"muse-voice-transcribe-1.0"},"mistral/voxtral-mini-transcribe-realtime-latest":{"mode":"audio_transcription","source":"https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-realtime-26-02","supported_endpoints":["/v1/audio/transcriptions"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"mistral","base_model":"voxtral-mini-transcribe-realtime"},"mistral/voxtral-mini-tts-latest":{"mode":"audio_speech","source":"https://docs.mistral.ai/models/model-cards/voxtral-tts-26-03","supported_endpoints":["/v1/audio/speech"],"supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_output":true,"provider":"mistral","base_model":"voxtral-mini-tts"},"mistral/mistral-ocr-4-0":{"mode":"ocr","supported_endpoints":["/v1/ocr","/v1/batch"],"source":"https://mistral.ai/pricing#api-pricing","provider":"mistral","base_model":"mistral-ocr-4-0"},"mistral/mistral-ocr-4-1":{"mode":"ocr","source":"https://docs.mistral.ai/models/model-cards/ocr-4-1","supported_endpoints":["/v1/ocr","/v1/batch"],"provider":"mistral","base_model":"mistral-ocr-4-1"},"mistral/mistral-ocr-2512":{"mode":"ocr","supported_endpoints":["/v1/ocr","/v1/batch"],"source":"https://mistral.ai/pricing#api-pricing","provider":"mistral","base_model":"mistral-ocr"},"parallel_ai/search-fast":{"mode":"search","provider":"parallel_ai","base_model":"search-fast"},"parallel_ai/search-turbo":{"mode":"search","provider":"parallel_ai","base_model":"search-turbo"},"reducto/parse-legacy":{"mode":"ocr","source":"https://reducto.ai/pricing","supported_endpoints":["/v1/ocr"],"provider":"reducto","base_model":"parse-legacy"},"reducto/parse-v3":{"mode":"ocr","source":"https://reducto.ai/pricing","supported_endpoints":["/v1/ocr"],"provider":"reducto","base_model":"parse-v3"},"rerank-v4.0-fast":{"max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"mode":"rerank","source":"https://cohere.com/pricing","provider":"cohere","base_model":"rerank-v4.0-fast"},"rerank-v4.0-pro":{"max_input_tokens":32768,"max_output_tokens":32768,"max_tokens":32768,"mode":"rerank","source":"https://cohere.com/pricing","provider":"cohere","base_model":"rerank-v4.0-pro"},"you_com/search":{"mode":"search","provider":"you_com","base_model":"search"},"transcribe/StartTranscriptionJob":{"mode":"audio_transcription","source":"https://aws.amazon.com/transcribe/pricing/","metadata":{"notes":"Amazon Transcribe standard batch transcription, billed per second of audio with no minimum. Same rate in every region of the AWS Price List offer file for transcribe (checked 2026-09-17)"},"provider":"transcribe","base_model":"starttranscriptionjob"},"vertex_ai/chirp_3":{"metadata":{"calculation":"$0.016/60 seconds = $0.00026667 per second","original_pricing_per_minute":0.016},"mode":"audio_transcription","source":"https://cloud.google.com/speech-to-text/pricing","supported_endpoints":["/v1/audio/transcriptions","/v1/realtime"],"provider":"vertex_ai","base_model":"chirp-3"},"vertex_ai/lyria-002":{"mode":"audio_speech","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#lyria","supported_audio_formats":["wav"],"supported_endpoints":["/v1/audio/speech"],"supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_output":true,"vertex_ai_audio_api":"lyria_predict","provider":"vertex_ai","base_model":"lyria-002"},"vertex_ai/lyria-3-clip-preview":{"max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"mode":"audio_speech","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#lyria","supported_audio_formats":["mp3"],"supported_endpoints":["/v1beta/interactions","/v1/audio/speech"],"supported_modalities":["text"],"supported_output_modalities":["audio"],"supported_regions":["global"],"supports_audio_input":false,"supports_audio_output":true,"supports_function_calling":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_vision":false,"supports_web_search":false,"vertex_ai_audio_api":"lyria_interactions","provider":"vertex_ai","base_model":"lyria-3-clip"},"vertex_ai/lyria-3-pro-preview":{"max_input_tokens":131072,"max_output_tokens":8192,"max_tokens":8192,"mode":"audio_speech","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#lyria","supported_audio_formats":["mp3","wav"],"supported_endpoints":["/v1beta/interactions","/v1/audio/speech"],"supported_modalities":["text"],"supported_output_modalities":["audio"],"supported_regions":["global"],"supports_audio_input":false,"supports_audio_output":true,"supports_function_calling":false,"supports_prompt_caching":false,"supports_response_schema":false,"supports_system_messages":false,"supports_vision":false,"supports_web_search":false,"vertex_ai_audio_api":"lyria_interactions","provider":"vertex_ai","base_model":"lyria-3-pro"},"vertex_ai/veo-3.1-lite-generate-001":{"max_input_tokens":1024,"max_tokens":1024,"mode":"video_generation","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing#veo","supported_modalities":["text","image"],"supported_output_modalities":["video"],"provider":"vertex_ai","base_model":"veo-3.1-lite-generate"},"voyage/rerank-3":{"max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"mode":"rerank","source":"https://docs.voyageai.com/docs/pricing","provider":"voyage","base_model":"rerank-3"},"voyage/rerank-3-lite":{"max_input_tokens":32000,"max_output_tokens":32000,"max_tokens":32000,"mode":"rerank","source":"https://docs.voyageai.com/docs/pricing","provider":"voyage","base_model":"rerank-3-lite"},"runwayml/gen4.5":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["video"],"metadata":{"comment":"12 credits per second @ $0.01 per credit = $0.12 per second. The API uses ratio values like 1280:720; converted here to WxH resolution strings."},"provider":"runwayml","base_model":"gen4.5","supported_resolutions":["1280x720","720x1280","1104x832","960x960","832x1104"]},"runwayml/aleph2":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","video"],"supported_output_modalities":["video"],"metadata":{"comment":"28 credits per second @ $0.01 per credit = $0.28 per second; 56 credit minimum per task not modeled"},"provider":"runwayml","base_model":"aleph2"},"runwayml/seedance2":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image","video"],"supported_output_modalities":["video"],"metadata":{"comment":"36 credits per second @ $0.01 per credit = $0.36 per second (480p/720p); 40 credits/s = $0.40 (1080p); 150 credits/s = $1.50 (4K). No credit minimum. Supports text-to-video, image-to-video, and video-to-video at the same per-second rate."},"provider":"runwayml","base_model":"seedance2"},"runwayml/seedance2_fast":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image","video"],"supported_output_modalities":["video"],"metadata":{"comment":"29 credits per second at 480p/720p @ $0.01 per credit = $0.29 per second"},"provider":"runwayml","base_model":"seedance2-fast"},"runwayml/seedance2_mini":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image","video"],"supported_output_modalities":["video"],"metadata":{"comment":"16 credits per second @ $0.01 per credit = $0.16 per second; 64 credit minimum per task not modeled"},"provider":"runwayml","base_model":"seedance2-mini"},"runwayml/seedance2_5":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image","video"],"supported_output_modalities":["video"],"metadata":{"comment":"Output: 20/30/68 credits per second at 480p/720p/1080p @ $0.01 per credit; input video billed additionally at 10/15/34 credits per input second and the 80 credit minimum per task are not modeled"},"provider":"runwayml","base_model":"seedance2-5"},"runwayml/hailuo3":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image","video"],"supported_output_modalities":["video"],"metadata":{"comment":"10 credits per second at 768P, 15 at 2K (mapped to the 1080p tier) @ $0.01 per credit; 2 credits per reference image not modeled"},"provider":"runwayml","base_model":"hailuo3"},"runwayml/gemini_omni_flash":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image","video"],"supported_output_modalities":["video"],"metadata":{"comment":"10 credits per second @ $0.01 per credit = $0.10 per second"},"provider":"runwayml","base_model":"gemini-omni-flash"},"runwayml/veo3.1":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["video"],"metadata":{"comment":"40 credits per second with audio, 20 without @ $0.01 per credit; priced at the with-audio rate"},"provider":"runwayml","base_model":"veo3.1"},"runwayml/veo3.1_fast":{"mode":"video_generation","source":"https://docs.dev.runwayml.com/guides/pricing/","supported_modalities":["text","image"],"supported_output_modalities":["video"],"metadata":{"comment":"15 credits per second with audio, 10 without @ $0.01 per credit; priced at the with-audio rate"},"provider":"runwayml","base_model":"veo3.1-fast"},"scaleway/openai/whisper-large-v3":{"mode":"audio_transcription","provider":"scaleway","base_model":"whisper-large-v3"},"gpt-realtime-whisper":{"mode":"audio_transcription","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime","/v1/realtime/transcription_sessions"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"openai","base_model":"gpt-realtime-whisper"},"gemini/gemini-3.1-flash-live-preview":{"max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"mode":"realtime","source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"rpm":10,"gemini_audio_only_live":true,"supports_response_schema":false,"provider":"gemini","base_model":"gemini-3.1-flash-live"},"gemini/gemini-3.1-flash-tts-preview":{"max_input_tokens":8192,"max_output_tokens":16384,"max_tokens":16384,"mode":"audio_speech","source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/audio/speech"],"tpm":4000000,"rpm":10,"supports_function_calling":false,"supports_response_schema":false,"supports_web_search":false,"provider":"gemini","base_model":"gemini-3.1-flash-tts"},"duckduckgo/search":{"mode":"search","metadata":{"notes":"DuckDuckGo Instant Answer API is free and does not require an API key."},"provider":"duckduckgo","base_model":"search"},"soniox/stt-async-v4":{"max_output_tokens":8000,"max_tokens":8000,"mode":"audio_transcription","source":"https://soniox.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"supports_audio_input":true,"provider":"soniox","base_model":"stt-async-v4"},"soniox/stt-async-v5":{"max_output_tokens":8000,"max_tokens":8000,"mode":"audio_transcription","source":"https://soniox.com/pricing","supported_endpoints":["/v1/audio/transcriptions"],"supports_audio_input":true,"provider":"soniox","base_model":"stt-async-v5"},"gpt-transcribe":{"mode":"audio_transcription","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/audio/transcriptions","/v1/realtime/transcription_sessions"],"supported_modalities":["audio","text"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"openai","base_model":"gpt-transcribe","supports_native_streaming":true},"gpt-live-transcribe":{"mode":"audio_transcription","source":"https://developers.openai.com/api/docs/pricing","supported_endpoints":["/v1/realtime","/v1/realtime/transcription_sessions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"openai","base_model":"gpt-live-transcribe"},"gpt-live-1":{"mode":"realtime","source":"https://developers.openai.com/api/docs/pricing","supported_modalities":["text","audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"provider":"openai","base_model":"gpt-live-1"},"gpt-realtime-translate":{"max_input_tokens":16000,"max_output_tokens":2000,"max_tokens":2000,"mode":"realtime","source":"https://developers.openai.com/api/docs/pricing","supported_modalities":["audio"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"provider":"openai","base_model":"gpt-realtime-translate"},"mistral/mistral-moderation-2603":{"max_input_tokens":131072,"mode":"moderation","source":"https://docs.mistral.ai/models/model-cards/mistral-moderation-26-03","provider":"mistral","base_model":"mistral-moderation"},"mistral/voxtral-mini-2602":{"mode":"audio_transcription","source":"https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02","supported_endpoints":["/v1/audio/transcriptions"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"mistral","base_model":"voxtral-mini"},"mistral/voxtral-mini-transcribe-realtime-2602":{"mode":"audio_transcription","source":"https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-realtime-26-02","supported_endpoints":["/v1/audio/transcriptions"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"mistral","base_model":"voxtral-mini-transcribe-realtime"},"mistral/voxtral-mini-tts-2603":{"mode":"audio_speech","source":"https://docs.mistral.ai/models/model-cards/voxtral-tts-26-03","supported_endpoints":["/v1/audio/speech"],"supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_output":true,"provider":"mistral","base_model":"voxtral-mini-tts"},"fallback_generalizations":{"rules":[{"name":"anthropic-claude-adaptive-thinking","pattern":"(?:opus|sonnet|haiku)[-._](?:4[-._](?:[6-9]|[1-9]\\d)(?!\\d)|(?:[5-9]|[1-9]\\d{1,})[-._]\\d{1,2}(?!\\d))","description":"Claude opus/sonnet/haiku at version 4.6 or higher: 4.6 through 4.99, then any 5.x, 6.x or later major. The minor is capped at two digits so an 8-digit date suffix such as claude-opus-4-20250514 is never read as a >= 4.6 minor. Turns on adaptive thinking for new families with no code change.","extends":"anthropic-claude","model_info":{"supports_adaptive_thinking":true}},{"name":"anthropic-claude","pattern":"^claude-[a-z]+-\\d+[-.]\\d+(?:-\\d{8})?$","description":"Any Claude family-major-minor id, optionally with an 8-digit date suffix, anchored to the whole name. Version-neutral fallback that gives an unmapped Claude provider routing and baseline capabilities; it carries no pricing, so cost stays on the standard unpriced behavior rather than a guessed number.","model_info":{"mode":"chat","max_input_tokens":200000,"max_output_tokens":64000,"max_tokens":64000,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_vision":true,"supports_tool_choice":true,"supports_assistant_prefill":true,"supports_prompt_caching":true,"supports_response_schema":true,"supports_reasoning":true,"supports_pdf_input":true,"supports_system_messages":true}}],"base_model":"fallback-generalizations"},"gemini/gemini-3.5-live-translate-preview":{"max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"mode":"realtime","rpm":10,"source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["audio"],"supported_output_modalities":["audio","text"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":false,"supports_response_schema":false,"supports_web_search":false,"tpm":250000,"provider":"gemini","base_model":"gemini-3.5-live-translate"},"gemini/gemini-3.5-transcribe":{"mode":"audio_transcription","source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/audio/transcriptions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"tpm":800000,"rpm":2000,"supports_function_calling":false,"provider":"gemini","base_model":"gemini-3.5-transcribe"},"gemini/gemini-3.5-transcribe-live":{"mode":"audio_transcription","source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"tpm":250000,"rpm":10,"supports_function_calling":false,"provider":"gemini","base_model":"gemini-3.5-transcribe-live"},"vertex_ai/gemini-3.5-transcribe-preview":{"mode":"audio_transcription","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/v1/audio/transcriptions"],"supported_modalities":["text","audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"vertex_ai","base_model":"gemini-3.5-transcribe"},"vertex_ai/gemini-3.5-transcribe-live-preview":{"mode":"audio_transcription","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"vertex_ai","base_model":"gemini-3.5-transcribe-live"},"vertex_ai/gemini-3.5-live-translate-preview":{"mode":"realtime","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["audio"],"supported_output_modalities":["audio","text"],"supports_audio_input":true,"supports_audio_output":true,"provider":"vertex_ai","base_model":"gemini-3.5-live-translate"},"xai/grok-imagine-image":{"mode":"image_generation","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/images/generations"],"supported_modalities":["text","image"],"supported_output_modalities":["image"],"provider":"xai","base_model":"grok-imagine-image"},"xai/grok-imagine-image-2026-03-02":{"mode":"image_generation","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/images/generations"],"supported_modalities":["text","image"],"supported_output_modalities":["image"],"provider":"xai","base_model":"grok-imagine-image"},"xai/grok-imagine-image-quality":{"deprecation_date":"2026-11-02","mode":"image_generation","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/images/generations"],"supported_modalities":["text","image"],"supported_output_modalities":["image"],"provider":"xai","base_model":"grok-imagine-image-quality"},"xai/grok-imagine-image-quality-20260403":{"deprecation_date":"2026-11-02","mode":"image_generation","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/images/generations"],"supported_modalities":["text","image"],"supported_output_modalities":["image"],"provider":"xai","base_model":"grok-imagine-image-quality"},"xai/grok-imagine-image-quality-latest":{"deprecation_date":"2026-11-02","mode":"image_generation","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/images/generations"],"supported_modalities":["text","image"],"supported_output_modalities":["image"],"provider":"xai","base_model":"grok-imagine-image-quality"},"xai/grok-imagine-image-2.0":{"mode":"image_generation","source":"https://docs.x.ai/docs/models","supported_endpoints":["/v1/images/generations"],"supported_modalities":["text","image"],"supported_output_modalities":["image"],"provider":"xai","base_model":"grok-imagine-image-2.0"},"xai/grok-imagine-video":{"mode":"video_generation","source":"https://docs.x.ai/docs/models/grok-imagine-video","supported_modalities":["text","image","video"],"supported_output_modalities":["video"],"provider":"xai","base_model":"grok-imagine-video"},"xai/grok-imagine-video-1.5":{"mode":"video_generation","source":"https://docs.x.ai/docs/models/grok-imagine-video-1.5","supported_modalities":["text","image","audio"],"supported_output_modalities":["video"],"provider":"xai","base_model":"grok-imagine-video-1.5"},"xai/grok-imagine-video-1.5-2026-05-30":{"mode":"video_generation","source":"https://docs.x.ai/docs/models/grok-imagine-video-1.5","supported_modalities":["text","image","audio"],"supported_output_modalities":["video"],"provider":"xai","base_model":"grok-imagine-video-1.5"},"xai/grok-imagine-video-1.5-preview":{"mode":"video_generation","source":"https://docs.x.ai/docs/models/grok-imagine-video-1.5","supported_modalities":["text","image","audio"],"supported_output_modalities":["video"],"provider":"xai","base_model":"grok-imagine-video-1.5"},"xai/grok-voice-transcribe-1.0":{"metadata":{"calculation":"$0.10/3600 seconds = $0.00002778 per second","original_pricing_per_hour":0.1},"mode":"audio_transcription","source":"https://docs.x.ai/developers/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"xai","base_model":"grok-voice-transcribe-1.0"},"xai/grok-voice-transcribe-2.0":{"metadata":{"calculation":"$0.10/3600 seconds = $0.00002778 per second","original_pricing_per_hour":0.1},"mode":"audio_transcription","source":"https://docs.x.ai/developers/pricing","supported_endpoints":["/v1/audio/transcriptions"],"provider":"xai","base_model":"grok-voice-transcribe-2.0"},"mistral/mistral-ocr-3":{"mode":"ocr","supported_endpoints":["/v1/ocr","/v1/batch"],"source":"https://mistral.ai/pricing#api-pricing","provider":"mistral","base_model":"mistral-ocr-3"},"mistral/mistral-ocr-3-0":{"mode":"ocr","supported_endpoints":["/v1/ocr","/v1/batch"],"source":"https://mistral.ai/pricing#api-pricing","provider":"mistral","base_model":"mistral-ocr-3-0"},"mistral/mistral-ocr-4":{"mode":"ocr","source":"https://docs.mistral.ai/models/model-cards/ocr-4-1","supported_endpoints":["/v1/ocr","/v1/batch"],"provider":"mistral","base_model":"mistral-ocr-4"},"mistral/voxtral-mini-latest":{"mode":"audio_transcription","source":"https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-26-02","supported_endpoints":["/v1/audio/transcriptions"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"mistral","base_model":"voxtral-mini"},"mistral/voxtral-mini-realtime-2602":{"mode":"audio_transcription","source":"https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-realtime-26-02","supported_endpoints":["/v1/audio/transcriptions"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"mistral","base_model":"voxtral-mini-realtime"},"mistral/voxtral-mini-realtime-latest":{"mode":"audio_transcription","source":"https://docs.mistral.ai/models/model-cards/voxtral-mini-transcribe-realtime-26-02","supported_endpoints":["/v1/audio/transcriptions"],"supported_modalities":["audio"],"supported_output_modalities":["text"],"supports_audio_input":true,"provider":"mistral","base_model":"voxtral-mini-realtime"},"elevenlabs/scribe_v2":{"mode":"audio_transcription","source":"https://elevenlabs.io/pricing/api","supported_endpoints":["/v1/audio/transcriptions"],"provider":"elevenlabs","base_model":"scribe-v2"},"cloudflare/@cf/openai/whisper":{"mode":"audio_transcription","rpm":720,"source":"https://developers.cloudflare.com/workers-ai/models/whisper/","supported_endpoints":["/v1/audio/transcriptions"],"provider":"cloudflare","base_model":""},"cloudflare/@cf/openai/whisper-large-v3-turbo":{"mode":"audio_transcription","rpm":720,"source":"https://developers.cloudflare.com/workers-ai/models/whisper-large-v3-turbo/","supported_endpoints":["/v1/audio/transcriptions"],"provider":"cloudflare","base_model":""},"vertex_ai/gemini-2.5-flash-native-audio":{"mode":"realtime","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","base_model":"gemini-2.5-flash-native-audio"},"vertex_ai/gemini-2.5-flash-preview-tts":{"mode":"audio_speech","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","base_model":"gemini-2.5-flash-tts"},"vertex_ai/gemini-3.1-flash-tts-preview":{"mode":"audio_speech","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","base_model":"gemini-3.1-flash-tts"},"vertex_ai/gemini-3.5-transcribe":{"mode":"audio_transcription","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","base_model":"gemini-3.5-transcribe"},"vertex_ai/gemini-3.5-transcribe-live":{"mode":"audio_transcription","source":"https://cloud.google.com/gemini-enterprise-agent-platform/generative-ai/pricing","provider":"vertex_ai","base_model":"gemini-3.5-transcribe-live"},"azure/eu/text-embedding-3-large":{"deprecation_date":"2028-02-09","mode":"embedding","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","base_model":"text-embedding-3-large"},"azure/eu/text-embedding-3-small":{"deprecation_date":"2028-02-09","mode":"embedding","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","base_model":"text-embedding-3-small"},"azure/eu/text-embedding-ada-002":{"deprecation_date":"2028-02-09","mode":"embedding","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","base_model":"text-embedding-ada-002"},"gemini/gemini-3.8-live":{"max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"mode":"realtime","source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_vision":true,"supports_web_search":true,"gemini_audio_only_live":true,"supports_response_schema":false,"provider":"gemini","base_model":"gemini-3.8-live"},"gemini/gemini-3.8-live-extended-thinking":{"max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"mode":"realtime","source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_vision":true,"supports_web_search":true,"gemini_audio_only_live":true,"supports_reasoning":true,"supports_response_schema":false,"provider":"gemini","base_model":"gemini-3.8-live-extended-thinking"},"azure/us/text-embedding-3-large":{"deprecation_date":"2028-02-09","mode":"embedding","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","base_model":"text-embedding-3-large"},"azure/us/text-embedding-3-small":{"deprecation_date":"2028-02-09","mode":"embedding","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","base_model":"text-embedding-3-small"},"azure/us/text-embedding-ada-002":{"deprecation_date":"2028-02-09","mode":"embedding","source":"https://prices.azure.com/api/retail/prices?$filter=serviceName%20eq%20'Foundry%20Models'%20and%20armRegionName%20eq%20'eastus'%20and%20priceType%20eq%20'Consumption'","provider":"azure","base_model":"text-embedding-ada-002"},"typesafe/jev-1.13.0":{"mode":"decisions","source":"https://docs.typesafe.ai/models","provider":"typesafe","base_model":"jev-1.13.0","max_input_tokens":65536,"max_tokens":65536,"rpm":1200,"tpm":15000000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_vision":false,"supports_audio_input":false,"supports_video_input":false,"supports_pdf_input":false,"supports_function_calling":false,"supports_tool_choice":false},"typesafe/jev-latest":{"mode":"decisions","source":"https://docs.typesafe.ai/models","provider":"typesafe","base_model":"jev-1.13.0","max_input_tokens":65536,"max_tokens":65536,"rpm":1200,"tpm":15000000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_vision":false,"supports_audio_input":false,"supports_video_input":false,"supports_pdf_input":false,"supports_function_calling":false,"supports_tool_choice":false},"typesafe/jev-preview":{"mode":"decisions","source":"https://docs.typesafe.ai/models","provider":"typesafe","base_model":"jev-1.13.0","max_input_tokens":65536,"max_tokens":65536,"rpm":1200,"tpm":15000000,"supported_modalities":["text"],"supported_output_modalities":["text"],"supports_vision":false,"supports_audio_input":false,"supports_video_input":false,"supports_pdf_input":false,"supports_function_calling":false,"supports_tool_choice":false},"replicate/google/imagen-4":{"mode":"image_generation","provider":"replicate","base_model":"imagen-4"},"replicate/bytedance/seedream-4":{"mode":"image_generation","provider":"replicate","base_model":"seedream-4"},"replicate/black-forest-labs/flux-2-max":{"mode":"image_generation","provider":"replicate","base_model":"flux-2-max"},"replicate/google/nano-banana-pro":{"mode":"image_generation","provider":"replicate","base_model":"nano-banana-pro"},"replicate/black-forest-labs/flux-2-flex":{"mode":"image_generation","provider":"replicate","base_model":"flux-2-flex"},"replicate/black-forest-labs/flux-2-pro":{"mode":"image_generation","provider":"replicate","base_model":"flux-2-pro"},"replicate/google/imagen-4-ultra":{"mode":"image_generation","provider":"replicate","base_model":"imagen-4-ultra"},"replicate/google/imagen-4-fast":{"mode":"image_generation","provider":"replicate","base_model":"imagen-4-fast"},"replicate/google/imagen-3":{"mode":"image_generation","provider":"replicate","base_model":"imagen-3"},"replicate/google/imagen-3-fast":{"mode":"image_generation","provider":"replicate","base_model":"imagen-3-fast"},"replicate/xai/grok-imagine-image":{"mode":"image_generation","provider":"replicate","base_model":"grok-imagine-image"},"replicate/google/nano-banana":{"mode":"image_generation","provider":"replicate","base_model":"nano-banana"},"replicate/qwen/qwen-image":{"mode":"image_generation","provider":"replicate","base_model":"qwen-image"},"replicate/black-forest-labs/flux-kontext-pro":{"mode":"image_generation","provider":"replicate","base_model":"flux-kontext-pro"},"replicate/black-forest-labs/flux-kontext-max":{"mode":"image_generation","provider":"replicate","base_model":"flux-kontext-max"},"replicate/black-forest-labs/flux-2-klein-4b":{"mode":"image_generation","provider":"replicate","base_model":"flux-2-klein-4b"},"replicate/black-forest-labs/flux-1.1-pro-ultra":{"mode":"image_generation","provider":"replicate","base_model":"flux-1.1-pro-ultra"},"replicate/black-forest-labs/flux-pro":{"mode":"image_generation","provider":"replicate","base_model":"flux-pro"},"replicate/stability-ai/stable-diffusion-3.5-medium":{"mode":"image_generation","provider":"replicate","base_model":"stability-ai/stable-diffusion-3.5-medium"},"replicate/stability-ai/stable-diffusion-3.5-large":{"mode":"image_generation","provider":"replicate","base_model":"stability-ai/stable-diffusion-3.5-large"},"replicate/black-forest-labs/flux-dev":{"mode":"image_generation","provider":"replicate","base_model":"flux-dev"},"replicate/black-forest-labs/flux-schnell":{"mode":"image_generation","provider":"replicate","base_model":"flux-schnell"},"replicate/prunaai/p-image":{"mode":"image_generation","provider":"replicate","base_model":"prunaai/p-image"},"replicate/bytedance/seedream-5-lite":{"mode":"image_generation","provider":"replicate","base_model":"seedream-5-lite"},"replicate/prunaai/hidream-l1-fast":{"mode":"image_generation","provider":"replicate","base_model":"prunaai/hidream-l1-fast"},"replicate/prunaai/flux-fast":{"mode":"image_generation","provider":"replicate","base_model":"prunaai/flux-fast"},"replicate/recraft-ai/recraft-v3":{"mode":"image_generation","provider":"replicate","base_model":"recraft-ai/recraft-v3"},"replicate/ideogram-ai/ideogram-v3-turbo":{"mode":"image_generation","provider":"replicate","base_model":"ideogram-ai/ideogram-v3-turbo"},"replicate/openai/gpt-image-1.5":{"mode":"image_generation","provider":"replicate","base_model":"gpt-image-1.5"},"replicate/google/veo-3":{"mode":"video_generation","provider":"replicate","base_model":"veo-3"},"replicate/google/veo-3-fast":{"mode":"video_generation","provider":"replicate","base_model":"veo-3-fast"},"replicate/google/veo-3.1":{"mode":"video_generation","provider":"replicate","base_model":"veo-3.1"},"replicate/google/veo-3.1-fast":{"mode":"video_generation","provider":"replicate","base_model":"veo-3.1-fast"},"replicate/google/veo-2":{"mode":"video_generation","provider":"replicate","base_model":"veo-2"},"replicate/bytedance/seedance-1-pro":{"mode":"video_generation","provider":"replicate","base_model":"seedance-1-pro"},"replicate/bytedance/seedance-1-lite":{"mode":"video_generation","provider":"replicate","base_model":"seedance-1-lite"},"replicate/minimax/hailuo-02":{"mode":"video_generation","provider":"replicate","base_model":"hailuo-02"},"replicate/minimax/hailuo-02-fast":{"mode":"video_generation","provider":"replicate","base_model":"hailuo-02-fast"},"replicate/minimax/video-01":{"mode":"video_generation","provider":"replicate","base_model":"video-01"},"replicate/minimax/video-01-live":{"mode":"video_generation","provider":"replicate","base_model":"video-01-live"},"replicate/minimax/video-01-director":{"mode":"video_generation","provider":"replicate","base_model":"video-01-director"},"replicate/kwaivgi/kling-v2.1-master":{"mode":"video_generation","provider":"replicate","base_model":"kwaivgi/kling-v2.1-master"},"replicate/kwaivgi/kling-v2.1":{"mode":"video_generation","provider":"replicate","base_model":"kwaivgi/kling"},"replicate/kwaivgi/kling-v2.0":{"mode":"video_generation","provider":"replicate","base_model":"kwaivgi/kling"},"replicate/kwaivgi/kling-v1.6-pro":{"mode":"video_generation","provider":"replicate","base_model":"kwaivgi/kling-v1.6-pro"},"replicate/kwaivgi/kling-v1.6-standard":{"mode":"video_generation","provider":"replicate","base_model":"kwaivgi/kling-v1.6-standard"},"replicate/leonardoai/motion-2.0":{"mode":"video_generation","provider":"replicate","base_model":"leonardoai/motion-2.0"},"replicate/runwayml/gen4-turbo":{"mode":"video_generation","provider":"replicate","base_model":"gen4-turbo"},"replicate/wan-video/wan-2.2-t2v-fast":{"mode":"video_generation","provider":"replicate","base_model":"wan-video/wan-2.2-t2v-fast"},"replicate/wan-video/wan-2.2-i2v-fast":{"mode":"video_generation","provider":"replicate","base_model":"wan-video/wan-2.2-i2v-fast"},"replicate/wan-video/wan-2.2-5b-fast":{"mode":"video_generation","provider":"replicate","base_model":"wan-video/wan-2.2-5b-fast"},"replicate/luma/ray-2-540p":{"mode":"video_generation","provider":"replicate","base_model":"luma/ray-2-540p"},"replicate/luma/ray-2-720p":{"mode":"video_generation","provider":"replicate","base_model":"luma/ray-2-720p"},"replicate/luma/ray-flash-2-540p":{"mode":"video_generation","provider":"replicate","base_model":"luma/ray-flash-2-540p"},"replicate/luma/ray-flash-2-720p":{"mode":"video_generation","provider":"replicate","base_model":"luma/ray-flash-2-720p"},"replicate/pixverse/pixverse-v4.5":{"mode":"video_generation","provider":"replicate","base_model":"pixverse/pixverse"},"replicate/pixverse/pixverse-v4":{"mode":"video_generation","provider":"replicate","base_model":"pixverse/pixverse-v4"},"bedrock/amazon.nova-canvas-v1:0":{"deprecation_date":"2026-09-30","max_input_tokens":2600,"mode":"image_generation","supports_nova_canvas_image_edit":true,"provider":"bedrock","base_model":"nova-canvas"},"stability.stable-diffusion-xl-v1":{"max_input_tokens":77,"max_tokens":77,"mode":"image_generation","provider":"bedrock","base_model":"stable-diffusion-xl-v1"},"stability.stable-diffusion-xl-v0":{"max_input_tokens":77,"max_tokens":77,"mode":"image_generation","provider":"bedrock","base_model":"stable-diffusion-xl-v0"},"azure/dall-e-3":{"mode":"image_generation","provider":"azure","base_model":"dall-e-3"},"openrouter/openai/gpt-image-2.5-sunburst":{"provider":"openrouter","base_model":"gpt-image-2.5-sunburst","mode":"image_generation","max_input_tokens":400000,"max_output_tokens":360000,"max_tokens":360000,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true},"openrouter/openai/gpt-image-2.5-flare":{"provider":"openrouter","base_model":"gpt-image-2.5-flare","mode":"image_generation","max_input_tokens":400000,"max_output_tokens":360000,"max_tokens":360000,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true},"openrouter/microsoft/mai-image-2.6":{"provider":"openrouter","base_model":"mai-image-2.6","mode":"image_generation","max_input_tokens":4096,"max_output_tokens":1024,"max_tokens":1024,"supports_vision":true},"openrouter/microsoft/mai-image-2.6-flash":{"provider":"openrouter","base_model":"mai-image-2.6-flash","mode":"image_generation","max_input_tokens":4096,"max_output_tokens":1024,"max_tokens":1024,"supports_vision":true},"openrouter/microsoft/mai-image-2.5-pro":{"provider":"openrouter","base_model":"mai-image-2.5-pro","mode":"image_generation","max_input_tokens":4096,"max_output_tokens":1024,"max_tokens":1024,"supports_vision":true},"openrouter/openai/gpt-image-2":{"provider":"openrouter","base_model":"gpt-image-2","mode":"image_generation","max_input_tokens":400000,"max_output_tokens":360000,"max_tokens":360000,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true},"openrouter/openai/gpt-image-1":{"provider":"openrouter","base_model":"gpt-image-1","mode":"image_generation","max_input_tokens":400000,"max_output_tokens":360000,"max_tokens":360000,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true},"openrouter/openai/gpt-image-1-mini":{"provider":"openrouter","base_model":"gpt-image-1-mini","mode":"image_generation","max_input_tokens":400000,"max_output_tokens":360000,"max_tokens":360000,"supports_response_schema":true,"supports_vision":true,"supports_prompt_caching":true},"openrouter/microsoft/mai-image-2.5":{"provider":"openrouter","base_model":"mai-image-2.5","mode":"image_generation","max_input_tokens":4096,"max_output_tokens":1024,"max_tokens":1024,"supports_vision":true},"replicate/bytedance/seedream-4.5":{"mode":"image_generation","provider":"replicate","base_model":"seedream-4.5"},"replicate/black-forest-labs/flux-1.1-pro":{"mode":"image_generation","provider":"replicate","base_model":"flux-1.1-pro"},"replicate/black-forest-labs/flux-2-klein-9b":{"mode":"image_generation","provider":"replicate","base_model":"flux-2-klein-9b"},"replicate/black-forest-labs/flux-2-klein-9b-base":{"mode":"image_generation","provider":"replicate","base_model":"flux-2-klein-9b-base"},"replicate/black-forest-labs/flux-krea-dev":{"mode":"image_generation","provider":"replicate","base_model":"flux-krea-dev"},"replicate/ideogram-ai/ideogram-v3-balanced":{"mode":"image_generation","provider":"replicate","base_model":"ideogram-v3-balanced"},"replicate/recraft-ai/recraft-v4.1":{"mode":"image_generation","provider":"replicate","base_model":"recraft-v4.1"},"replicate/recraft-ai/recraft-v4.1-utility":{"mode":"image_generation","provider":"replicate","base_model":"recraft-v4.1-utility"},"replicate/xai/grok-imagine-image-quality":{"mode":"image_generation","provider":"replicate","base_model":"grok-imagine-image-quality"},"replicate/xai/grok-imagine-video":{"mode":"video_generation","provider":"replicate","base_model":"grok-imagine-video"},"replicate/recraft-ai/recraft-vectorize":{"mode":"image_generation","provider":"replicate","base_model":"recraft-vectorize"},"replicate/bytedance/seedance-2.0":{"mode":"video_generation","provider":"replicate","base_model":"seedance-2.0"},"replicate/kwaivgi/kling-v3-video":{"mode":"video_generation","provider":"replicate","base_model":"kling-v3-video"},"replicate/minimax/hailuo-2.3":{"mode":"video_generation","provider":"replicate","base_model":"hailuo-2.3"},"replicate/wan-video/wan-2.7-t2v":{"mode":"video_generation","provider":"replicate","base_model":"wan-2.7-t2v"},"replicate/wan-video/wan-2.6-t2v":{"mode":"video_generation","provider":"replicate","base_model":"wan-2.6-t2v"},"replicate/pixverse/pixverse-v6":{"mode":"video_generation","provider":"replicate","base_model":"pixverse-v6"},"replicate/tencent/hunyuan-3d-3.1":{"mode":"3d","provider":"replicate","base_model":"hunyuan-3d-3.1"},"replicate/qwen/qwen-image-edit-2511":{"mode":"image_generation","provider":"replicate","base_model":"qwen-image-edit-2511"},"replicate/qwen/qwen-edit-multiangle":{"mode":"image_generation","provider":"replicate","base_model":"qwen-edit-multiangle"},"azure/gpt-6-sol-2026-09-22":{"base_model":"gpt-6-sol","provider":"azure","mode":"chat","max_input_tokens":922000,"max_output_tokens":128000,"supports_vision":true,"supports_image_input":true,"supported_output_modalities":["text"],"supports_reasoning":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_native_streaming":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_none_reasoning_effort":true,"source":"https://azure.microsoft.com/en-us/blog/gpt-6-astra-sol-and-luna-for-production-agents-in-microsoft-foundry/"},"azure/us/gpt-6-sol-2026-09-22":{"base_model":"gpt-6-sol","provider":"azure","mode":"chat","max_input_tokens":922000,"max_output_tokens":128000,"supports_vision":true,"supports_image_input":true,"supported_output_modalities":["text"],"supports_reasoning":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_native_streaming":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_none_reasoning_effort":true,"source":"https://azure.microsoft.com/en-us/blog/gpt-6-astra-sol-and-luna-for-production-agents-in-microsoft-foundry/"},"azure/eu/gpt-6-sol-2026-09-22":{"base_model":"gpt-6-sol","provider":"azure","mode":"chat","max_input_tokens":922000,"max_output_tokens":128000,"supports_vision":true,"supports_image_input":true,"supported_output_modalities":["text"],"supports_reasoning":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_native_streaming":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_none_reasoning_effort":true,"source":"https://azure.microsoft.com/en-us/blog/gpt-6-astra-sol-and-luna-for-production-agents-in-microsoft-foundry/"},"azure/gpt-6-luna-2026-09-22":{"base_model":"gpt-6-luna","provider":"azure","mode":"chat","max_input_tokens":922000,"max_output_tokens":128000,"supports_vision":true,"supports_image_input":true,"supported_output_modalities":["text"],"supports_reasoning":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_native_streaming":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_none_reasoning_effort":true,"source":"https://azure.microsoft.com/en-us/blog/gpt-6-astra-sol-and-luna-for-production-agents-in-microsoft-foundry/"},"azure/us/gpt-6-luna-2026-09-22":{"base_model":"gpt-6-luna","provider":"azure","mode":"chat","max_input_tokens":922000,"max_output_tokens":128000,"supports_vision":true,"supports_image_input":true,"supported_output_modalities":["text"],"supports_reasoning":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_native_streaming":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_none_reasoning_effort":true,"source":"https://azure.microsoft.com/en-us/blog/gpt-6-astra-sol-and-luna-for-production-agents-in-microsoft-foundry/"},"azure/eu/gpt-6-luna-2026-09-22":{"base_model":"gpt-6-luna","provider":"azure","mode":"chat","max_input_tokens":922000,"max_output_tokens":128000,"supports_vision":true,"supports_image_input":true,"supported_output_modalities":["text"],"supports_reasoning":true,"supports_function_calling":true,"supports_parallel_function_calling":true,"supports_native_streaming":true,"supports_native_structured_output":true,"supports_response_schema":true,"supports_none_reasoning_effort":true,"source":"https://azure.microsoft.com/en-us/blog/gpt-6-astra-sol-and-luna-for-production-agents-in-microsoft-foundry/"},"together_ai/google/imagen-4.0-preview":{"provider":"together_ai","base_model":"imagen-4.0-preview","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/google/imagen-4.0-fast":{"provider":"together_ai","base_model":"imagen-4.0-fast","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/google/imagen-4.0-ultra":{"provider":"together_ai","base_model":"imagen-4.0-ultra","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/google/flash-image-2.5":{"provider":"together_ai","base_model":"flash-image-2.5","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/google/gemini-3-pro-image":{"provider":"together_ai","base_model":"gemini-3-pro-image","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/black-forest-labs/FLUX.1.1-pro":{"provider":"together_ai","base_model":"flux.1.1-pro","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/black-forest-labs/FLUX.1-kontext-pro":{"provider":"together_ai","base_model":"flux.1-kontext-pro","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/black-forest-labs/FLUX.1-kontext-max":{"provider":"together_ai","base_model":"flux.1-kontext-max","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/black-forest-labs/FLUX.2-pro":{"provider":"together_ai","base_model":"flux.2-pro","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/black-forest-labs/FLUX.2-dev":{"provider":"together_ai","base_model":"flux.2-dev","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/black-forest-labs/FLUX.2-flex":{"provider":"together_ai","base_model":"flux.2-flex","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/ByteDance-Seed/Seedream-3.0":{"provider":"together_ai","base_model":"seedream-3.0","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/ByteDance-Seed/Seedream-4.0":{"provider":"together_ai","base_model":"seedream-4.0","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/ByteDance/Seedream-5.0-lite":{"provider":"together_ai","base_model":"seedream-5.0-lite","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/Qwen/Qwen-Image":{"provider":"together_ai","base_model":"qwen-image","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/RunDiffusion/Juggernaut-pro-flux":{"provider":"together_ai","base_model":"juggernaut-pro-flux","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/Rundiffusion/Juggernaut-Lightning-Flux":{"provider":"together_ai","base_model":"juggernaut-lightning-flux","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/ideogram/ideogram-3.0":{"provider":"together_ai","base_model":"ideogram-3.0","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/stabilityai/stable-diffusion-xl-base-1.0":{"provider":"together_ai","base_model":"stable-diffusion-xl-base-1.0","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/black-forest-labs/FLUX.2-max":{"provider":"together_ai","base_model":"flux.2-max","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/google/flash-image-3.1":{"provider":"together_ai","base_model":"flash-image-3.1","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/openai/gpt-image-1.5":{"provider":"together_ai","base_model":"gpt-image-1.5","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/Qwen/Qwen-Image-2.0":{"provider":"together_ai","base_model":"qwen-image-2.0","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/Qwen/Qwen-Image-2.0-Pro":{"provider":"together_ai","base_model":"qwen-image-2.0-pro","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/Wan-AI/Wan2.6-image":{"provider":"together_ai","base_model":"wan2.6-image","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/ideogram/ideogram-4.0":{"provider":"together_ai","base_model":"ideogram-4.0","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/openai/gpt-image-2":{"provider":"together_ai","base_model":"gpt-image-2","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/google/flash-image-3.1-lite":{"provider":"together_ai","base_model":"flash-image-3.1-lite","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/prunaai/p-image-ideogram":{"provider":"together_ai","base_model":"p-image-ideogram","mode":"image_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/minimax/video-01-director":{"provider":"together_ai","base_model":"video-01-director","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/minimax/hailuo-02":{"provider":"together_ai","base_model":"hailuo-02","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/google/veo-2.0":{"provider":"together_ai","base_model":"veo-2.0","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/ByteDance/Seedance-1.0-pro":{"provider":"together_ai","base_model":"seedance-1.0-pro","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/pixverse/pixverse-v5":{"provider":"together_ai","base_model":"pixverse-v5","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/kwaivgI/kling-2.1-master":{"provider":"together_ai","base_model":"kling-2.1-master","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/kwaivgI/kling-2.1-standard":{"provider":"together_ai","base_model":"kling-2.1-standard","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/kwaivgI/kling-2.1-pro":{"provider":"together_ai","base_model":"kling-2.1-pro","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/kwaivgI/kling-1.6-standard":{"provider":"together_ai","base_model":"kling-1.6-standard","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/vidu/vidu-q1":{"provider":"together_ai","base_model":"vidu-q1","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/openai/sora-2":{"provider":"together_ai","base_model":"sora-2","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/openai/sora-2-pro":{"provider":"together_ai","base_model":"sora-2-pro","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/ByteDance/Seedance-1.0-lite":{"provider":"together_ai","base_model":"seedance-1.0-lite","mode":"video_generation","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/canopylabs/orpheus-3b-0.1-ft":{"provider":"together_ai","base_model":"orpheus-3b-0.1-ft","mode":"audio_speech","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/hexgrad/Kokoro-82M":{"provider":"together_ai","base_model":"kokoro-82m","mode":"audio_speech","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/cartesia/sonic-3":{"provider":"together_ai","base_model":"sonic-3","mode":"audio_speech","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/cartesia/sonic-2":{"provider":"together_ai","base_model":"sonic-2","mode":"audio_speech","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/cartesia/sonic":{"provider":"together_ai","base_model":"sonic","mode":"audio_speech","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/openai/whisper-large-v3":{"provider":"together_ai","base_model":"whisper-large-v3","mode":"audio_transcription","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/nvidia/parakeet-tdt-0.6b-v3":{"provider":"together_ai","base_model":"parakeet-tdt-0.6b-v3","mode":"audio_transcription","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/nvidia/nemotron-3-asr-streaming-0.6b":{"provider":"together_ai","base_model":"nemotron-3-asr-streaming-0.6b","mode":"audio_transcription","source":"https://docs.together.ai/docs/serverless/models.md"},"together_ai/nvidia/nemotron-3.5-asr-streaming-0.6b":{"provider":"together_ai","base_model":"nemotron-3.5-asr-streaming-0.6b","mode":"audio_transcription","source":"https://docs.together.ai/docs/serverless/models.md"},"vertex_ai/gemini-2.5-flash-tts":{"base_model":"gemini-2.5-flash-tts","mode":"audio_speech","source":"https://models.dev/api.json","max_input_tokens":32768,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_output":true,"supports_function_calling":false,"supports_reasoning":false,"provider":"vertex_ai"},"vertex_ai/gemini-2.5-pro-tts":{"base_model":"gemini-2.5-pro-tts","mode":"audio_speech","source":"https://models.dev/api.json","max_input_tokens":32768,"max_output_tokens":16384,"max_tokens":16384,"supported_modalities":["text"],"supported_output_modalities":["audio"],"supports_audio_output":true,"supports_function_calling":false,"supports_reasoning":false,"provider":"vertex_ai"},"replicate/topazlabs/image-upscale":{"mode":"image_generation","provider":"replicate","base_model":"image-upscale"},"replicate/prunaai/p-image-upscale":{"mode":"image_generation","provider":"replicate","base_model":"prunaai/p-image-upscale"},"gemini-3.1-flash-live-preview":{"max_input_tokens":131072,"max_output_tokens":65536,"max_tokens":65536,"mode":"realtime","source":"https://ai.google.dev/gemini-api/docs/pricing","supported_endpoints":["/v1/realtime"],"supported_modalities":["text","image","audio","video"],"supported_output_modalities":["text","audio"],"supports_audio_input":true,"supports_audio_output":true,"supports_function_calling":true,"supports_vision":true,"supports_web_search":true,"tpm":250000,"rpm":10,"gemini_audio_only_live":true,"supports_response_schema":false,"provider":"gemini","base_model":"gemini-3.1-flash-live"},"claude-3-7-sonnet-20250219":{"deprecation_date":"2026-02-19","is_deprecated":true},"claude-3-opus-20240229":{"deprecation_date":"2026-01-05","is_deprecated":true},"claude-opus-4-20250514":{"deprecation_date":"2026-06-15","is_deprecated":true}}